<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1629794</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>SaGP: identifying plant saline-alkali tolerance genes based on machine learning techniques</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Qiao</surname>
<given-names>Baixue</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2805666/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Gao</surname>
<given-names>Wentao</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Xudong</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Du</surname>
<given-names>Min</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Shuda</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Xuanrui</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3083308/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pang</surname>
<given-names>Shaozi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Chunxue</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Jiang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhao</surname>
<given-names>Yuming</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1262661/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xie</surname>
<given-names>Linan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1547530/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Ecology, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Key Laboratory of Saline-Alkali Vegetation Ecology Restoration, Ministry of Education, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>State Key Laboratory of Tree Genetics and Breeding, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>College of Computer and Control Engineering, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>College of Landscape Architecture, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>College of Life Science, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Key Laboratory of Sustainable Forest Ecosystem Management-Ministry of Education, School of Ecology, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Lijun Dou, Cleveland Clinic, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Jun Yan, China Agricultural University, China</p>
<p>Zilong Zhang, Hainan University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Jiang Wang, <email xlink:href="mailto:wj-312@126.com">wj-312@126.com</email>; Yuming Zhao, <email xlink:href="mailto:zym@nefu.edu.cn">zym@nefu.edu.cn</email>; Linan Xie, <email xlink:href="mailto:linanxie@nefu.edu.cn">linanxie@nefu.edu.cn</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work and share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1629794</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Qiao, Gao, Zhang, Du, Wang, Liu, Pang, Yang, Wang, Zhao and Xie</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Qiao, Gao, Zhang, Du, Wang, Liu, Pang, Yang, Wang, Zhao and Xie</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Mining novel genes underlying agronomical traits is a crucial subject in plant biology, essential for enhancing crop quality, ensuring food security, and preserving biodiversity. Wet experiments are the main methods to uncover genes with target functions but are expensive and time-consuming. Machine learning, in contrast, can accelerate the gene discovery process by learning from accumulated data, making it more efficient and cost-effective. However, despite their potential, existing machine-learning tools to mine stress-resistant genes in plants are scarce. In this study, we developed the first known machine learning model, SaGP (Saline-alkali Genes Prediction), to identify plant saline-alkali tolerance genes based on sequencing data. It outperformed traditional computational tools, <italic>i.e.</italic>, BLAST, and correctly identified the latest published genes. Moreover, we utilized SaGP to evaluate three recently published genes: <italic>GhAG2</italic>, <italic>MdBPR6</italic>, and <italic>TaCCD1</italic>. SaGP correctly identified all their functions. Overall, these results suggest that SaGP can be used for the large-scale identification of saline-alkali tolerance genes and served as a framework for the development of additional automated tools, thus promoting crop breeding and plant conservation. To efficiently identify salt-alkali resistant genes in large-scale data, we developed a user-friendly, freely accessible web service platform based on SaGP (<ext-link ext-link-type="uri" xlink:href="https://www.sagprediction.com/">https://www.sagprediction.com/</ext-link>).</p>
</abstract>
<kwd-group>
<kwd>machine learning</kwd>
<kwd>saline-alkali tolerance genes</kwd>
<kwd>gene mining</kwd>
<kwd>feature selection</kwd>
<kwd>SAGP</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="54"/>
<page-count count="10"/>
<word-count count="4055"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Functional and Applied Plant Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Enhancing plants&#x2019; tolerance to abiotic stresses has long been the focus in biology and breeding science. Early efforts focused purely on plant phenotypes (<xref ref-type="bibr" rid="B34">Meuwissen et&#xa0;al., 2001</xref>; <xref ref-type="bibr" rid="B35">Meyer et&#xa0;al., 2012</xref>). Later works began to decipher the genetic bases underlying key traits based on quantitative trait loci (QTL) mapping (<xref ref-type="bibr" rid="B25">Kang et&#xa0;al., 2019</xref>) and genome-wide association studies (GWAS) (<xref ref-type="bibr" rid="B21">Gupta, 2021</xref>). Many functional genetic variants have been identified, resulting in breeding plants with excellent traits (<xref ref-type="bibr" rid="B45">Wang et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B54">Zhao et&#xa0;al., 2024</xref>) and developing effective species conservation strategies (<xref ref-type="bibr" rid="B8">Chen et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B20">Gougherty et&#xa0;al., 2021</xref>). However, despite these achievements, these works are time-consuming and costly (<xref ref-type="bibr" rid="B17">Dou et&#xa0;al., 2021</xref>) and overall have low precision in determining functional variants (<xref ref-type="bibr" rid="B33">Mackay et&#xa0;al., 2009</xref>; <xref ref-type="bibr" rid="B46">Wray et&#xa0;al., 2013</xref>), resulting in inefficient plant selection and breeding. Moreover, they focus on a few model species, such as Arabidopsis, maize, rice, etc. Taking full advantage of knowledge from these species and utilizing them to boost the identification of functional genetic variants in other species are still challenging.</p>
<p>With the development of genetics and informatics, a new framework is now being proposed to boost efficiency and cut the cost of current research, i.e., Breeding 4.0 (<xref ref-type="bibr" rid="B43">Wallace et&#xa0;al., 2018</xref>). It is characterized by high-throughput sequence data (<xref ref-type="bibr" rid="B15">Ding et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B49">Yang et&#xa0;al., 2013</xref>) combined with computational methods (<xref ref-type="bibr" rid="B23">Jarrahi, 2018</xref>). Traditional computational methods, such as BLAST (<xref ref-type="bibr" rid="B1">Altschul et&#xa0;al., 1990</xref>), may fit Breeding 4.0, but their poor accuracy can lead to inefficiency (<xref ref-type="bibr" rid="B12">Dai et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B29">Li et&#xa0;al., 2022</xref>). On the other hand, machine learning (ML) may provide an alternative to traditional computational approaches (<xref ref-type="bibr" rid="B19">Fu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B37">Qiao et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B42">Van Dijk et&#xa0;al., 2021</xref>). It has been used in genomic selection-assisted breeding (<xref ref-type="bibr" rid="B48">Yan and Wang, 2023</xref>) and in assessing plants&#x2019; vulnerability under future climates regarding their genetic compositions (<xref ref-type="bibr" rid="B38">Sang et&#xa0;al., 2022</xref>). Moreover, several studies have implemented machine learning algorithms to identify plant genes with specific functions. For example, PGB was used to detect photosynthetic-related genes based on a voting algorithm (<xref ref-type="bibr" rid="B44">Wang et&#xa0;al., 2022</xref>). DRPPP based on SVM was created to predict disease resistance proteins with high performances (<xref ref-type="bibr" rid="B36">Pal et&#xa0;al., 2016</xref>). ConSReg based on regularized LASSO was developed to identify key transcription factors responsive to specific abiotic stresses, which outperformed traditional enrichment-based methods (<xref ref-type="bibr" rid="B39">Song et&#xa0;al., 2020</xref>). These works can provide important tools for Breeding 4.0 to precisely screen target genes on a large scale and thus facilitate crop improvement and species conservation. Unfortunately, similar work to identify genes resistant to abiotic stresses are scarce.</p>
<p>In this study, we proposed a framework to construct intelligent tools to identify novel plant abiotic stress resistant genes. Focusing on saline-alkali stress, i.e., excessive accumulation of neutral salts and sodic salts that leads to decreased crop productivity (<xref ref-type="bibr" rid="B22">James et&#xa0;al., 2012</xref>) and the loss of native biodiversity worldwide (<xref ref-type="bibr" rid="B7">Briggs and Taws, 2003</xref>), we developed the first known machine learning model (SaGP, Saline-alkali Genes Prediction) to identify plant saline-alkali tolerance genes. It achieves 0.99 prediction accuracy better than BLAST assessed with the independent test dataset. To further evaluate the performance of SaGP, we tested some latest published genes, including <italic>GhAG2</italic> (<xref ref-type="bibr" rid="B50">Yu et&#xa0;al., 2022</xref>), <italic>MdBPR6</italic> (<xref ref-type="bibr" rid="B52">Zhang et&#xa0;al., 2023</xref>), and <italic>TaCCD1</italic> (<xref ref-type="bibr" rid="B11">Cui et&#xa0;al., 2023</xref>), and SaGP correctly identified all their functions. Overall, the results suggest that SaGP can be used to fast and accurately identify saline-alkali tolerance genes in plants on a large scale with sequencing data, thus promoting crop breeding and plant conservation. SaGP is freely available at <ext-link ext-link-type="uri" xlink:href="https://www.sagprediction.com">www.sagprediction.com</ext-link>.</p>
</sec>
<sec id="s2" sec-type="results">
<label>2</label>
<title>Results</title>
<sec id="s2_1">
<label>2.1</label>
<title>Model comparison and SaGP construction</title>
<p>The five cost-sensitive methods performed differently regarding their capacity to distinguish saline-alkali tolerance and non-tolerance genes. Overall, the Weighted Cross-Entropy (WCE) method had the best performances regarding MCC, Balanced Accuracy, and PR-AUC (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>; see <xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figure S1</bold>
</xref> for Accuracy, F1 score, and ROC-AUC values). Moreover, different groups of features showed different pertinency to the gene function of saline-alkali tolerance. Among them, ACC-PSSM achieved the highest and most stable performances across all metrics, followed by PDT-Profile (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>; <xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figure S1</bold>
</xref>). In contrast, several features, such as ACC and AAAFF, had the lowest performances (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>; <xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figure S1</bold>
</xref>). We further compared the performances of SaGP models constructed using four different feature sets&#x2014;ACC-PSSM, all features, and the top two and top five groups ranked by MCC&#x2014;with those of traditional tools HMMER and BLAST. The performances of BLAST were generally low with respect to MCC, Balanced Accuarcy, F1 score, and Accuracy (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2a</bold>
</xref>). Only 84% and 3.6% of saline-alkali tolerance genes can be correctly identified by BLAST (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2a</bold>
</xref>), respectively, suggesting their inability to screen for saline-alkali tolerance genes on a large scale. The set with all features performed slightly better than ACC-PSSM in terms of MCC, while ACC-PSSM performed best regarding F1 score and Balanced Accuracy (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2a</bold>
</xref>). Because extracting all features is time-consuming, we thus implemented SaGP based on ACC-PSSM.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The performances of five cost-sensitive methods and 40 groups of protein features based on the test dataset. EL, Equalization loss; FL, Focal loss; LaL, Logit-adjusted loss; LdaML, Label-distribution-aware margin loss; WCE, weighted cross-entropy; MCC, Matthew&#x2019;s Correlation Coefficient; PR-AUC, the area under the precision-recall. See <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> for the details of 40 groups of protein features.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1629794-g001.tif">
<alt-text content-type="machine-generated">A graph showing three plots: MCC, Balanced Accuracy, and PR AUC, with performance metrics for different models. Symbols represent various models: triangles for EL, stars for FL, diamonds for LaL, crosses for LdaML, and circles for WCE. X-axis lists model names; Y-axis shows respective performance scores, with lines and markers indicating results across models.</alt-text>
</graphic>
</fig>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>
<bold>(a)</bold> The performances of ACC-PSSM, all features, MCC2, MCC5, BLAST and HMMER based on the test dataset. <bold>(b)</bold> Performance comparison of SVM, RF, XGBoost, DNN, and SaGP on the independent test dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1629794-g002.tif">
<alt-text content-type="machine-generated">Circular bar charts labeled &#x201c;a&#x201d; and &#x201c;b&#x201d; showing performance metrics for BLAST, HMMER, ACC_PSSM. Panel &#x201c;a&#x201d; displays metrics for MCC, Balanced Accuracy, F1 score, and Accuracy using colored segments. Panel &#x201c;b&#x201d; presents radial bars with values for MCC, Balanced Accuracy, PR-AUC, ROC-AUC for SaGP, RF, XGBoost, SVM, and DNN, using shades of blue.</alt-text>
</graphic>
</fig>
<p>Next, we compared the classification performances of the SaGP with the other four classifiers&#x2014;SVM, Random Forest (RF), XGBoost, and deep neural network (DNN)&#x2014;under the same cost-sensitive learning setting using the WCE loss function. The comparison was based on five evaluation metrics: Accuracy, F1 score, Area Under the ROC Curve (AUROC), Area Under the Precision-Recall Curve (AUPRC), and MCC. Among all models, SaGP outperformed all other classifiers, achieving the highest MCC (0.5988) and AUPRC (0.6021) (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2b</bold>
</xref>), which underscores its superior ability to correctly identify saline-alkali tolerance genes under imbalanced conditions. It also attained competitive values in F1 score (0.5563) and AUROC (0.9408) (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2b</bold>
</xref>), indicating both reliable classification and strong ranking capability.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Performance of SVM, RF, XGBoost, DNN, and SaGP on the independent test dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="left">Accuracy</th>
<th valign="top" align="left">Balanced Accuracy</th>
<th valign="top" align="left">F1</th>
<th valign="top" align="left">ROC-AUC</th>
<th valign="top" align="left">PR-AUC</th>
<th valign="top" align="left">MCC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SaGP</td>
<td valign="top" align="left">0.989 &#xb1; 0.0006</td>
<td valign="top" align="left">0.649 &#xb1; 0.0226</td>
<td valign="top" align="left">0.556 &#xb1; 0.0688</td>
<td valign="top" align="left">0.941 &#xb1; 0.0261</td>
<td valign="top" align="left">0.602 &#xb1; 0.0651</td>
<td valign="top" align="left">0.598 &#xb1; 0.0600</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">0.987 &#xb1; 0.0005</td>
<td valign="top" align="left">0.556 &#xb1; 0.0147</td>
<td valign="top" align="left">0.200 &#xb1; 0.0473</td>
<td valign="top" align="left">0.814 &#xb1; 0.0265</td>
<td valign="top" align="left">0.368 &#xb1; 0.0669</td>
<td valign="top" align="left">0.328 &#xb1; 0.0452</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.988 &#xb1; 0.0007</td>
<td valign="top" align="left">0.604 &#xb1; 0.0256</td>
<td valign="top" align="left">0.341 &#xb1; 0.0694</td>
<td valign="top" align="left">0.882 &#xb1; 0.0281</td>
<td valign="top" align="left">0.493 &#xb1; 0.0762</td>
<td valign="top" align="left">0.448 &#xb1; 0.0558</td>
</tr>
<tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">0.987 &#xb1; 0.0012</td>
<td valign="top" align="left">0.574 &#xb1; 0.0361</td>
<td valign="top" align="left">0.441 &#xb1; 0.0842</td>
<td valign="top" align="left">0.835 &#xb1; 0.0351</td>
<td valign="top" align="left">0.445 &#xb1; 0.0683</td>
<td valign="top" align="left">0.503 &#xb1; 0.0750</td>
</tr>
<tr>
<td valign="top" align="left">DNN</td>
<td valign="top" align="left">0.99 &#xb1; 0.0011</td>
<td valign="top" align="left">0.710 &#xb1; 0.0333</td>
<td valign="top" align="left">0.554 &#xb1; 0.0645</td>
<td valign="top" align="left">0.883 &#xb1; 0.0300</td>
<td valign="top" align="left">0.529 &#xb1; 0.0632</td>
<td valign="top" align="left">0.583 &#xb1; 0.0592</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To further evaluate the capacity of SaGP to identify novel saline-alkali tolerance genes, we predicted the three latest published genes, i.e., <italic>GhAG2</italic> (<xref ref-type="bibr" rid="B50">Yu et&#xa0;al., 2022</xref>), <italic>MdBPR6</italic> (<xref ref-type="bibr" rid="B52">Zhang et&#xa0;al., 2023</xref>), and <italic>TaCCD1</italic> (<xref ref-type="bibr" rid="B11">Cui et&#xa0;al., 2023</xref>). The predictions were consistent with the experimental results in the literature (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>), supporting that SaGP can correctly uncover novel saline-alkali tolerance genes.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>The 40 groups of protein features extracted in our study, their abbreviations, and corresponding tools.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Gene</th>
<th valign="top" align="left">SaGP Prediction</th>
<th valign="top" align="left">Experiment</th>
<th valign="top" align="left">Description</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">
<italic>GhAG2</italic>
</td>
<td valign="top" align="left">yes</td>
<td valign="top" align="left">salt resistance</td>
<td valign="top" align="left">In cotton, the over-expression of <italic>GhAG2</italic> increased the germination rate under the saline environment <break/>(<xref ref-type="bibr" rid="B50">Yu et&#xa0;al., 2022</xref>).</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>MdBPR6</italic>
</td>
<td valign="top" align="left">yes</td>
<td valign="top" align="left">salt sensitivity</td>
<td valign="top" align="left">In apple, suppression of <italic>MdPRP6</italic> reduces the accumulation of ROS and Na<sup>+</sup> under the saline environment (<xref ref-type="bibr" rid="B52">Zhang et&#xa0;al., 2023</xref>).</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>TaCCD1</italic>
</td>
<td valign="top" align="left">yes</td>
<td valign="top" align="left">alkali sensitivity</td>
<td valign="top" align="left">In wheat, suppression of <italic>TaCCD1</italic> can promote plant growth under the alkaline environment <break/>(<xref ref-type="bibr" rid="B11">Cui et&#xa0;al., 2023</xref>).</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Feature importance analysis</title>
<p>We next analyzed the contribution of individual ACC-PSSM features to the SaGP. Based on gain values, the top 20 most important features were identified (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3a</bold>
</xref>). Features such as ACC_PSSM_F3215, ACC_PSSM_F2649, and ACC_PSSM_F137 contributed most to the model&#x2019;s performance. Correlation analysis revealed low redundancy among these features, with most pairwise Pearson correlation coefficients below 0.5 (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3b</bold>
</xref>), indicating they capture distinct aspects of the input data. SHAP value analysis further confirmed the importance and directionality of these features (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3c</bold>
</xref>). For example, higher values of ACC_PSSM_F3215 and ACC_PSSM_F2649 were positively associated with model output, suggesting their strong influence in identifying tolerant genes.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>
<bold>(a)</bold> Top 20 most important ACC_PSSM features based on gain. <bold>(b)</bold> Pairwise pearson correlation of top 20 ACC_PSSM features ranked by importance. <bold>(c)</bold> SHAP value summary for top ACC_PSSM features in SaGP.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1629794-g003.tif">
<alt-text content-type="machine-generated">a. Bar chart showing feature importance with ACC_PSSM_F3215 having the highest value above 10,000, followed by other features in descending order. b. Heatmap displaying correlation between features, ranging from red (high correlation) to blue (low correlation). c. SHAP summary plot illustrating the impact of features on model output, with feature values ranging from high (red) to low (blue).</alt-text>
</graphic>
</fig>
<p>To further explore their biological relevance, we investigated the potential functional significance of key features. ACC-PSSM_3215 This feature represents the autocovariance of proline residues at a lag of 9 within the PSSM (Position-Specific Scoring Matrix), capturing the evolutionary correlation between prolines separated by nine amino acid positions in the sequence. Proline is a well-established osmoprotectant in plants under salt stress, known for enhancing osmotic adjustment, stabilizing proteins and membrane structures, and mitigating oxidative damage through reactive oxygen species (ROS) scavenging. SHAP analysis revealed a positive association between higher values of this feature and the likelihood of a sequence being classified as a positive (salt-tolerant) sample. Notably, this feature exhibited significantly elevated values in salt-tolerant sequences, suggesting an enrichment of long-range proline interactions potentially involved in the formation of adaptive structural motifs or regulatory elements. These findings indicate that the model effectively captured biologically meaningful signals associated with proline-mediated stress adaptation. Importantly, despite the absence of explicit structural domain annotations, the model implicitly leveraged functional characteristics embedded within the primary sequence. The biological relevance of this proline-related feature thus provides strong support for both the predictive consistency of the positive samples and the interpretability of the model.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Web services of SaGP</title>
<p>To maximize the accessibility of the SaGP and minimize the difficulty of its use, we implemented it as a highly automated webserver (<ext-link ext-link-type="uri" xlink:href="https://www.sagprediction.com">https://www.sagprediction.com/</ext-link>) with JavaScript, Nodejs, Tailwind CSS (responsive design), HTML5, Docker, and Nginx. The only input from the users is the protein sequences encoded by their interested genes. SaGP will automatically process the sequences and return its predictions in a formatted table. Users are allowed to download the predicted results for future use.</p>
</sec>
</sec>
<sec id="s3" sec-type="discussion">
<label>3</label>
<title>Discussion</title>
<p>Deciphering gene functions has long been the central topic in biology and bioinformatics. With the advancement of high-throughput sequencing technologies, the massive accumulation of new sequences in public databases has far exceeded the capacity of traditional wet experiments. This has led to the development of computational methods and tools to accelerate the process of gene function identification, providing guidance for wet lab experiments and reducing the costs and time associated with wet experiments. One such method is homolog-based or domain-based (e.g., BLAST), involving comparing the genomic sequences of different organisms to infer gene functions based on their similarity to known genes. Another method is machine learning to predict the functions of unknown genes based on their sequence features. Several studies have compared the performance of both methods in identifying proteins with targeted functions, such as pathogenic proteins (<xref ref-type="bibr" rid="B29">Li et&#xa0;al., 2022</xref>) and antifreeze proteins (<xref ref-type="bibr" rid="B18">Eslami et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B24">Kandaswamy et&#xa0;al., 2011</xref>). Overall, these studies suggest that machine learning-based methods are superior to homolog/domain-based methods regarding speed and accuracy. Consistently, in this study, we found that the performances of SaGP were higher than homolog/domain-based methods.</p>
<p>One possible explanation for the incapacity of homolog/domain-based methods to identify salt-alkali tolerance genes may be caused by the fast protein evolution. In plants, the main mechanisms of salt-alkali tolerance involve ions transport (e.g., Na<sup>+</sup> and Ca<sup>2+</sup>) and detoxification (<xref ref-type="bibr" rid="B13">Deinlein et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B53">Zhang et&#xa0;al., 2022</xref>). Proteins with these biochemical and cellular functions tend to evolve more rapidly, resulting in low sequence similarities among homologous proteins (<xref ref-type="bibr" rid="B14">Devos and Valencia, 2000</xref>; <xref ref-type="bibr" rid="B37">Qiao et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B47">Xie et&#xa0;al., 2024</xref>) which disadvantages homolog-based and domain-based methods. Moreover, the functional space of genes/proteins is more complex than the sequence space, making it even more challenging to identify genes with specific functions based solely on sequence similarity (<xref ref-type="bibr" rid="B14">Devos and Valencia, 2000</xref>). SaGP, on the other hand, has the potential to overcome this challenge by capturing complex relations hidden in the sequence data based on machine learning algorithms and key protein features. Indeed, among all features, we found that ACC-PSSM performed best followed by PDT-Profile. Both ACC-PSSM and PDT-Profile capture evolutionary information (<xref ref-type="bibr" rid="B16">Dong et&#xa0;al., 2009</xref>; <xref ref-type="bibr" rid="B32">Liu et&#xa0;al., 2012</xref>). In addition, they also include sequence order effects (<xref ref-type="bibr" rid="B16">Dong et&#xa0;al., 2009</xref>; <xref ref-type="bibr" rid="B32">Liu et&#xa0;al., 2012</xref>), which may include information about local interactions/structures that are important for ion binding and transporting. Combining both groups of features barely improved model performances, suggesting that redundant information exists between them. Nevertheless, these results suggest evolution and sequence order are two crucial components for building machine learning tools to distinguish salt-alkali tolerance and non-tolerance genes in plants.</p>
<p>It is important to note that SaGP was trained with negative samples from <italic>Arabidopsis thaliana</italic>. It may have low performance to identify salt-alkali non-tolerance genes in species phylogenetically far distant from <italic>Arabidopsis thaliana</italic>. To evaluate the model&#x2019;s generalization capability across different species, we selected three latest published genes for validation: <italic>GhAG2</italic> (cotton) (<xref ref-type="bibr" rid="B50">Yu et&#xa0;al., 2022</xref>), <italic>MdBPR6</italic> (apple) (<xref ref-type="bibr" rid="B52">Zhang et&#xa0;al., 2023</xref>), and <italic>TaCCD1</italic> (wheat) (<xref ref-type="bibr" rid="B11">Cui et&#xa0;al., 2023</xref>). The prediction results of SaGP were consistent with the experimental results, indicating the effectiveness of SaGP in predicting salt-alkali resistant genes across different species. The significant advancements in sequencing technologies allows us to access extensive genetic data from a variety of plants more quickly and at a lower cost. However, due to the long growth cycles and high costs, the stress tolerance genes of many plants are not well studied. The application of SaGP provides superior guidance compared to BLAST for the rapid and accurate identification of salt-alkali tolerance genes in genomic data of these plants. Additionally, to efficiently identify salt-alkali resistant genes, we developed a user-friendly and freely accessible web service platform based on SaGP. This platform allows users to obtain prediction results only by inputting protein sequences, without the need for downloading models, installing software, or deploying any environment. In summary, SaGP offers a reliable identification tool for mining novel salt-alkali tolerant genes based on large-scale data, it also can serve as a fundamental model for the development of additional automated tools, which can greatly facilitate studies in plant genetics (<xref ref-type="bibr" rid="B30">Li et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B51">Zafar et&#xa0;al., 2024</xref>) and crop breeding (<xref ref-type="bibr" rid="B27">Kumar et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B40">Sun et&#xa0;al., 2023</xref>), and promoting global agricultural sustainability (<xref ref-type="bibr" rid="B40">Sun et&#xa0;al., 2023</xref>). As the availability of genomic data continues to grow, the expansion of the training dataset will further enhance predictive capabilities of SaGP.</p>
</sec>
<sec id="s4" sec-type="materials|methods">
<label>4</label>
<title>Materials and methods</title>
<sec id="s4_1">
<label>4.1</label>
<title>Data collection and processing</title>
<sec id="s4_1_1">
<label>4.1.1</label>
<title>Positive samples</title>
<p>Saline-alkali tolerance genes were manually curated from published literature. A total of 537 experimentally validated genes from 308 gene families were collected. To reduce potential confounding factors, transcription factors were removed. Additionally, we filtered out&#xa0;sequences containing irregular characters (e.g., &#x201c;X&#x201d;) and sequences shorter than 78 amino acids&#x2014;the minimum observed length among positive samples. To minimize sequence redundancy, we applied CD-HIT with a sequence identity threshold of 90%, resulting in 262 high-confidence non-redundant tolerance-related protein sequences.</p>
</sec>
<sec id="s4_1_2">
<label>4.1.2</label>
<title>Negative samples</title>
<p>Negative samples were collected from the Arabidopsis thaliana genome, specifically from the TAIR database (<xref ref-type="bibr" rid="B4">Berardini et&#xa0;al., 2015</xref>), after excluding any gene families known to be associated with saline-alkali tolerance. Transcription factors and low-quality sequences (containing non-standard residues or shorter than 78 amino acids) were also removed. CD-HIT (<xref ref-type="bibr" rid="B28">Li et&#xa0;al., 2001</xref>) was used to eliminate redundant sequences at a 90% identity threshold, yielding 17,753 non-tolerance protein sequences.</p>
<p>To further ensure the reliability of the negative dataset, we reanalyzed RNA-seq data from <xref ref-type="bibr" rid="B3">Anderson et&#xa0;al. (2018)</xref> (<xref ref-type="bibr" rid="B3">Anderson et&#xa0;al., 2018</xref>) (GEO accession: GSE116332), which profiled gene expression in Arabidopsis thaliana under both control and salt stress conditions. Gene expression levels were quantified using StringTie, and differential expression analysis was conducted using DESeq2. Notably, none of the negative genes exhibited significant differential expression between salt-treated and control conditions, confirming their non-responsiveness to salt stress at the transcriptomic level.</p>
</sec>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Feature extraction and selection</title>
<p>Engineering protein features to capture the underlying patterns of salt-alkali tolerance and non-tolerance genes is crucial to constructing accurate SaGP models (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>). Here, we used three programs to extract protein features, i.e., Pse-in-one2.0 (<xref ref-type="bibr" rid="B31">Liu et&#xa0;al., 2015</xref>), ftrCOOL (<xref ref-type="bibr" rid="B2">Amerifar et&#xa0;al., 2022</xref>), and MathFeature (<xref ref-type="bibr" rid="B6">Bonidia et&#xa0;al., 2022</xref>). Overall, 40 groups of protein features were extracted, representing important information about protein evolution, physicochemical properties, global and local sequence patterns, and residue interactions (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). To reduce computational complexity and feature redundancy, features with zero variance or highly correlated with other features (absolute Pearson correlation coefficients &gt; 0.8) were removed. The sequences data was split into training, validation, and independent test datasets with a ratio of 80:10:10. A univariate feature selection algorithm based on t-test and the training dataset was then used to select the set of features to construct machine learning models (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>). In total, 5377 features were retained.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>The framework of SaGP. In brief, protein sequences were used to construct SaGP. Both positive and negative sequences were validated with literature and RNA-seq data, respectively. Sequences were filtered to remove errors and redundancy. Protein features were extracted, selected, and used to train the machine learning models. The model with the best performances were evaluated based on the test dataset and were used to identify novel salt-alkali tolerance genes in plants.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1629794-g004.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a machine learning pipeline for protein sequence analysis. The process starts with data collection and processing, extracting features from positive and negative protein samples. In model building, selected features undergo training, evaluation, and testing, with 90% for training and 10% for testing. The model's performance is shown in a graph comparing predictions to actual results. Model prediction uses novel protein sequences, extracting features and generating predictions.</alt-text>
</graphic>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>The 40 groups of protein features extracted in our study, their abbreviations, and corresponding tools.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Features</th>
<th valign="middle" align="left">Tools</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Auto-cross covariance (ACC)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">Physicochemical distance transformation (PDT)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">Profile-based Auto-cross covariance (ACC-PSSM)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">PseAAC of Distance-Pairs and reduced alphabet scheme (Distance Pair)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">Distance-based Residue (DR)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">Profile-based physicochemical distance transformation (PDT-Profile)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">General parallel correlation pseudo amino acid composition (PC-PseAAC-General)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">Parallel correlation pseudo amino acid composition (PC-PseAAC)</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">Top-n-gram</td>
<td valign="middle" align="left">Pse-in-One 2.0</td>
</tr>
<tr>
<td valign="middle" align="left">Accumulated Amino Acid Frequency (AAAF)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Accumulated Amino Acid Frequency with Fourier (AAAFF)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Electron-ion interaction potential Mapping (EIIP Mapping)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Integer Mapping</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Kmer Frequency Mapping (KFM)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Amino acid composition (AAC)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Complex Networks without threshold</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Dipeptide composition (DPC)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Xmer k-Spaced Ymer Composition Frequency (kGap)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Tripeptide composition (TPC)</td>
<td valign="middle" align="left">MathFeature</td>
</tr>
<tr>
<td valign="middle" align="left">Amino Acid to K Part Composition (AAKC)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Amino Acid Autocorrelation-Autocovariance (AAutoCor)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Amphiphilic Pseudo-Amino Acid Composition(series) (APAAC)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Adaptive skip dipeptide composition (ASDC)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Composition of k-Spaced Grouped Amino Acids pairs (CkSGAApair)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Conjoint Triad (conjointTriad)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">k-Spaced Conjoint Triad (conjointTriadKS)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Composition_Transition_Distribution (CTD)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">CTD Composition (CTDC)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">CTD Distribution (CTDD)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">CTD Transition (CTDT)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Dipeptide Deviation from Expected Mean value (DDE)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Expected Value for each Amino Acid (ExpectedValueAA)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Expected Value for Grouped Amino Acid (ExpectedValueGAA)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Expected Value for Grouped K-mer Amino Acid (ExpectedValueGKmerAA)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Expected Value for K-mer Amino Acid (ExpectedValueKmerAA)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Grouped Amino Acid K Part Composition (GAAKpartComposition)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">k Grouped Amino Acid Composition (kGAAComposition)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Pseudo-Amino Acid Composition (Parallel) (PSEAAC)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Pseudo K_tuple Reduced Amino Acid Composition Type-11 (PseKRAAC_T11)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
<tr>
<td valign="middle" align="left">Quasi Sequence Order (QSOrder)</td>
<td valign="middle" align="left">ftrCOOL</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>SaGP construction and evaluation</title>
<p>Overall, the ratio between saline-alkali tolerance and non-tolerance sequences was 1:68, which leads to an imbalanced learning problem. To address this issue and to construct SaGP with potentially optimal performance, we tested the performances of cost-sensitive methods to tackle the imbalanced learning problem (<xref ref-type="bibr" rid="B41">Tanha et&#xa0;al., 2020</xref>) (<xref ref-type="bibr" rid="B5">Boldini et&#xa0;al., 2022</xref>). compared the potentials of five cost-sensitive methods, weighted cross-entropy (WCE), Focal loss (FL), Logit-adjusted loss (LaL), Label-distribution-aware margin loss (LdaML), and Equalization loss (EL), to improve imbalanced classification in drug discovery. Here, we followed their scheme to train and evaluate our models that is Lightgbm (<xref ref-type="bibr" rid="B26">Ke et&#xa0;al., 2017</xref>) was used to train the models based on the training dataset; each model was optimized using Hyperopt (<xref ref-type="bibr" rid="B5">Boldini et&#xa0;al., 2022</xref>) based on the validation dataset; their performances were evaluated based on the independent test dataset. To assess the relative importance of different group features for identifying saline-alkali tolerance and non-tolerance genes, we evaluated each group&#x2019;s features separately.</p>
<p>To comprehensively evaluate model performances, six metrics were calculated, i.e., Accuracy, Balanced Accuracy, F1 score, the area under the receiver operating characteristic curve (ROC-AUC), the area under the precision-recall (PR-AUC) curve and Matthew&#x2019;s Correlation Coefficient (MCC). Several studies have compared the performances of these metrics for imbalanced binary classification, and in general, MCC was recommended (<xref ref-type="bibr" rid="B9">Chicco and Jurman, 2020</xref>, <xref ref-type="bibr" rid="B10">2023</xref>). We, therefore, used MCC as the main reference to select the optimal model for SaGP.</p>
<p>To further confirm the power of SaGP to uncover novel saline-alkali tolerance genes, we collected three more genes from the latest publications, i.e., <italic>GhAG2</italic> (<xref ref-type="bibr" rid="B50">Yu et&#xa0;al., 2022</xref>), <italic>MdBPR6</italic> (<xref ref-type="bibr" rid="B52">Zhang et&#xa0;al., 2023</xref>), and <italic>TaCCD1</italic> (<xref ref-type="bibr" rid="B11">Cui et&#xa0;al., 2023</xref>), as additional tests. In addition, we assessed the performance of BLAST to identify salt-alkali tolerance genes. In brief, all salt-alkali tolerance genes in the training and validation datasets were used to construct the search database. Sequences from the test dataset were used as the query sequence of BLAST. E value 0.01 was used to indicate a significant similarity (hit).</p>
</sec>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <uri xlink:href="https://figshare.com/">https://figshare.com/</uri>, <uri xlink:href="https://figshare.com/account/home#/data">https://figshare.com/account/home#/data</uri>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>BQ: Visualization, Writing &#x2013; original draft. WG: Writing &#x2013; original draft, Visualization. XZ: Writing &#x2013; original draft, Data curation. MD: Data curation, Writing &#x2013; original draft. SW: Visualization, Writing &#x2013; original draft.. XL: Writing &#x2013; review &amp; editing, Data curation. SP: Writing &#x2013; review &amp; editing, Data curation. CY: Writing &#x2013; review &amp; editing. JW: Writing &#x2013; review &amp; editing. YZ: Writing &#x2013; review &amp; editing, Conceptualization. LX: Conceptualization, Supervision, Funding acquisition, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The authors declare that financial support was received for the research and/or publication of this article. This work was supported by the National Key R&amp;D Program of China during the 14th Five-Year Plan Period (Grant No. 2021YFD2200103), and the National Natural Science Foundation of China (Grant Nos. 62272094 and 62471123).</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2025.1629794/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2025.1629794/full#supplementary-material</ext-link>.</p>
<supplementary-material xlink:href="Image1.tif" id="SF1" mimetype="image/tiff">
<label>Supplementary Figure&#xa0;1</label>
<caption>
<p>The performances of five cost-sensitive methods and 40 groups of protein features based on the test dataset. Abbreviations: EL, Equalization loss; FL, Focal loss; LaL: Logit-adjusted loss; LdaML, Label-distribution-aware margin loss; WCE, weighted cross-entropy; MCC, Matthew&#x2019;s Correlation Coefficient; ROC-AUC, the area under the receiver operating characteristic curve. See <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> for the details of 40 groups of protein features.</p>
</caption>
</supplementary-material>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname> <given-names>S. F.</given-names>
</name>
<name>
<surname>Gish</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Miller</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Myers</surname> <given-names>E. W.</given-names>
</name>
<name>
<surname>Lipman</surname> <given-names>D. J.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Basic local alignment search tool</article-title>. <source>J. Mol. Biol.</source> <volume>215</volume>, <fpage>403</fpage>&#x2013;<lpage>410</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/s0022-2836(05)80360-2</pub-id>, PMID: <pub-id pub-id-type="pmid">2231712</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amerifar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Norouzi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ghandi</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A tool for feature extraction from biological sequences</article-title>. <source>Brief Bioinform.</source> <volume>23</volume>, <elocation-id>bbac108</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbac108</pub-id>, PMID: <pub-id pub-id-type="pmid">35383372</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anderson</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Kramer</surname> <given-names>M. C.</given-names>
</name>
<name>
<surname>Gosai</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Vandivier</surname> <given-names>L. E.</given-names>
</name>
<name>
<surname>Nelson</surname> <given-names>A. D. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>N(6)-methyladenosine inhibits local ribonucleolytic cleavage to stabilize mRNAs in arabidopsis</article-title>. <source>Cell Rep.</source> <volume>25</volume>, <fpage>1146</fpage>&#x2013;<lpage>1157.e3</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.celrep.2018.10.020</pub-id>, PMID: <pub-id pub-id-type="pmid">30380407</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berardini</surname> <given-names>T. Z.</given-names>
</name>
<name>
<surname>Reiser</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Mezheritsky</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Muller</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Strait</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>The Arabidopsis information resource: Making and mining the "gold standard" annotated reference plant genome</article-title>. <source>Genesis</source> <volume>53</volume>, <fpage>474</fpage>&#x2013;<lpage>485</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/dvg.22877</pub-id>, PMID: <pub-id pub-id-type="pmid">26201819</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boldini</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Friedrich</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Kuhn</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Sieber</surname> <given-names>S. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Tuning gradient boosting for imbalanced bioassay modelling with custom loss functions</article-title>. <source>J. Cheminform</source> <volume>14</volume>, <fpage>80</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13321-022-00657-w</pub-id>, PMID: <pub-id pub-id-type="pmid">36357942</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bonidia</surname> <given-names>R. P.</given-names>
</name>
<name>
<surname>Domingues</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>Sanches</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>de Carvalho</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MathFeature: feature extraction package for DNA, RNA and protein sequences based on mathematical descriptors</article-title>. <source>Brief Bioinform.</source> <volume>23</volume>, <elocation-id>bbab434</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbab434</pub-id>, PMID: <pub-id pub-id-type="pmid">34750626</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Briggs</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Taws</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Impacts of salinity on biodiversity&#x2014;clear understanding or muddy confusion</article-title>? <source>Aust. J. Bot.</source> <volume>51</volume>, <fpage>609</fpage>&#x2013;<lpage>617</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1071/BT02114</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Ericson</surname> <given-names>P. G. P.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>The combination of genomic offset and niche modelling provides insights into climate change-driven vulnerability</article-title>. <source>Nat. Commun.</source> <volume>13</volume>, <fpage>4821</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-022-32546-z</pub-id>, PMID: <pub-id pub-id-type="pmid">35974023</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chicco</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Jurman</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The advantages of the Matthews correlation coefficient (MCC) over F1 score and accuracy in binary classification evaluation</article-title>. <source>BMC Genomics</source> <volume>21</volume>, <elocation-id>6</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12864-019-6413-7</pub-id>, PMID: <pub-id pub-id-type="pmid">31898477</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chicco</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Jurman</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The Matthews correlation coefficient (MCC) should replace the ROC AUC as the standard metric for assessing binary classification</article-title>. <source>BioData Min</source> <volume>16</volume>, <elocation-id>4</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13040-023-00322-4</pub-id>, PMID: <pub-id pub-id-type="pmid">36800973</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cui</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Ca(2+)-dependent TaCCD1 cooperates with TaSAUR215 to enhance plasma membrane H(+)-ATPase activity and alkali stress tolerance by inhibiting PP2C-mediated dephosphorylation of TaHA2 in wheat</article-title>. <source>Mol. Plant</source> <volume>16</volume>, <fpage>571</fpage>&#x2013;<lpage>587</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molp.2023.01.010</pub-id>, PMID: <pub-id pub-id-type="pmid">36681864</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Tu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Schnable</surname> <given-names>J. C.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Non-homology-based prediction of gene functions in maize (Zea mays ssp. mays)</article-title>. <source>Plant Genome</source> <volume>13</volume>, <elocation-id>e20015</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/tpg2.20015</pub-id>, PMID: <pub-id pub-id-type="pmid">33016608</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deinlein</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Stephan</surname> <given-names>A. B.</given-names>
</name>
<name>
<surname>Horie</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Schroeder</surname> <given-names>J. I.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Plant salt-tolerance mechanisms</article-title>. <source>Trends Plant Sci.</source> <volume>19</volume>, <fpage>371</fpage>&#x2013;<lpage>379</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.tplants.2014.02.001</pub-id>, PMID: <pub-id pub-id-type="pmid">24630845</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devos</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Valencia</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Practical limits of function prediction</article-title>. <source>Proteins</source> <volume>41</volume>, <fpage>98</fpage>&#x2013;<lpage>107</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/1097-0134(20001001)41:1&lt;98::AID-PROT120&gt;3.0.CO;2-S</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>PlantCADB: A comprehensive plant chromatin accessibility database</article-title>. <source>Genomics Proteomics Bioinf.</source> <volume>21</volume>, <fpage>311</fpage>&#x2013;<lpage>323</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gpb.2022.10.005</pub-id>, PMID: <pub-id pub-id-type="pmid">36328151</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Guan</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A new taxonomy-based protein fold recognition approach based on autocross-covariance transformation</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>2655</fpage>&#x2013;<lpage>2662</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btp500</pub-id>, PMID: <pub-id pub-id-type="pmid">19706744</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dou</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A comprehensive review of the imbalance classification of protein post-translational modifications</article-title>. <source>Brief Bioinform.</source> <volume>22</volume>, <elocation-id>bbab089</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbab089</pub-id>, PMID: <pub-id pub-id-type="pmid">33834199</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eslami</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shirali Hossein Zade</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Takalloo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Mahdevar</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Emamjomeh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sajedi</surname> <given-names>R. H.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>afpCOOL: A tool for antifreeze protein prediction</article-title>. <source>Heliyon</source> <volume>4</volume>, <elocation-id>e00705</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.heliyon.2018.e00705</pub-id>, PMID: <pub-id pub-id-type="pmid">30094375</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Photonic machine learning with on-chip diffractive optics</article-title>. <source>Nat. Commun.</source> <volume>14</volume>, <fpage>70</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-022-35772-7</pub-id>, PMID: <pub-id pub-id-type="pmid">36604423</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gougherty</surname> <given-names>A. V.</given-names>
</name>
<name>
<surname>Keller</surname> <given-names>S. R.</given-names>
</name>
<name>
<surname>Fitzpatrick</surname> <given-names>M. C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Maladaptation, migration and extirpation fuel climate change risk in a forest tree species</article-title>. <source>Nat.&#xa0;Climate Change</source> <volume>11</volume>, <fpage>166</fpage>&#x2013;<lpage>171</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41558-020-00968-6</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname> <given-names>P. K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Quantitative genetics: pan-genomes, SVs, and k-mers for GWAS</article-title>. <source>Trends Genet.</source> <volume>37</volume>, <fpage>868</fpage>&#x2013;<lpage>871</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.tig.2021.05.006</pub-id>, PMID: <pub-id pub-id-type="pmid">34183185</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>James</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Blake</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zwart</surname> <given-names>A. B.</given-names>
</name>
<name>
<surname>Hare</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Rathjen</surname> <given-names>A. J.</given-names>
</name>
<name>
<surname>Munns</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Impact of ancestral wheat sodium exclusion genes Nax1 and Nax2 on grain yield of durum wheat on saline soils</article-title>. <source>Funct. Plant Biol.</source> <volume>39</volume>, <fpage>609</fpage>&#x2013;<lpage>618</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1071/fp12121</pub-id>, PMID: <pub-id pub-id-type="pmid">32480813</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jarrahi</surname> <given-names>M. H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Artificial intelligence and the future of work: Human-AI symbiosis in organizational decision making</article-title>. <source>Business Horizons</source> <volume>61</volume>, <fpage>577</fpage>&#x2013;<lpage>586</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.bushor.2018.03.007</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kandaswamy</surname> <given-names>K. K.</given-names>
</name>
<name>
<surname>Chou</surname> <given-names>K. C.</given-names>
</name>
<name>
<surname>Martinetz</surname> <given-names>T.</given-names>
</name>
<name>
<surname>M&#xf6;ller</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Suganthan</surname> <given-names>P. N.</given-names>
</name>
<name>
<surname>Sridharan</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>AFP-Pred: A random forest approach for predicting antifreeze proteins from sequence-derived properties</article-title>. <source>J. Theor. Biol.</source> <volume>270</volume>, <fpage>56</fpage>&#x2013;<lpage>62</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jtbi.2010.10.037</pub-id>, PMID: <pub-id pub-id-type="pmid">21056045</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname> <given-names>D. Y.</given-names>
</name>
<name>
<surname>Cheon</surname> <given-names>K. S.</given-names>
</name>
<name>
<surname>Oh</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Oh</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>S. L.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Rice genome resequencing reveals a major quantitative trait locus for resistance to bakanae disease caused by fusarium fujikuroi</article-title>. <source>Int. J. Mol. Sci.</source> <volume>20</volume>, <fpage>2598</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ijms20102598</pub-id>, PMID: <pub-id pub-id-type="pmid">31137840</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ke</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Finley</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>W.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Lightgbm: A&#xa0;highly efficient gradient boosting decision tree</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>, <fpage>3146</fpage>&#x2013;<lpage>3154</lpage>.</citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Choudhary</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Halder</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Prakash</surname> <given-names>N. R.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>V.</given-names>
</name>
<name>
<surname>V</surname> <given-names>V. T.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Salinity stress tolerance and omics approaches: revisiting the progress and achievements in major cereal crops</article-title>. <source>Heredity (Edinb)</source> <volume>128</volume>, <fpage>497</fpage>&#x2013;<lpage>518</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41437-022-00516-2</pub-id>, PMID: <pub-id pub-id-type="pmid">35249098</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Jaroszewski</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Godzik</surname> <given-names>A</given-names>
</name>
</person-group>. (<year>2001</year>). <article-title>Clustering of highly homologous sequences to reduce the size of large protein databases</article-title>. <source>Bioinformatics</source> <volume>17</volume>, <fpage>282</fpage>&#x2013;<lpage>283</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/17.3.282</pub-id>, PMID: <pub-id pub-id-type="pmid">11294794</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xiang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Pitt</surname> <given-names>M. E.</given-names>
</name>
<name>
<surname>Bainomugisa</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Coin</surname> <given-names>L. J. M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Computational analysis and prediction of PE_PGRS proteins using machine learning</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>20</volume>, <fpage>662</fpage>&#x2013;<lpage>674</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.csbj.2022.01.019</pub-id>, PMID: <pub-id pub-id-type="pmid">35140886</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Shao</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Rengel</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Transcriptome analysis reveals the molecular mechanisms underlying the enhancement of salt-tolerance in Melia azedarach under salinity stress</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>10981</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-024-61907-5</pub-id>, PMID: <pub-id pub-id-type="pmid">38745099</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chou</surname> <given-names>K. C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Pse-in-One: a web server for generating various modes of pseudo components of DNA, RNA, and protein sequences</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume>, <fpage>W65</fpage>&#x2013;<lpage>W71</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkv458</pub-id>, PMID: <pub-id pub-id-type="pmid">25958395</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Lan</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Using amino acid physicochemical distance transformation for fast protein remote homology detection</article-title>. <source>PloS One</source> <volume>7</volume>, <elocation-id>e46633</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0046633</pub-id>, PMID: <pub-id pub-id-type="pmid">23029559</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mackay</surname> <given-names>T. F.</given-names>
</name>
<name>
<surname>Stone</surname> <given-names>E. A.</given-names>
</name>
<name>
<surname>Ayroles</surname> <given-names>J. F.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>The genetics of quantitative traits: challenges and prospects</article-title>. <source>Nat. Rev. Genet.</source> <volume>10</volume>, <fpage>565</fpage>&#x2013;<lpage>577</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrg2612</pub-id>, PMID: <pub-id pub-id-type="pmid">19584810</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meuwissen</surname> <given-names>T. H.</given-names>
</name>
<name>
<surname>Hayes</surname> <given-names>B. J.</given-names>
</name>
<name>
<surname>Goddard</surname> <given-names>M. E.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Prediction of total genetic value using genome-wide dense marker maps</article-title>. <source>Genetics</source> <volume>157</volume>, <fpage>1819</fpage>&#x2013;<lpage>1829</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/genetics/157.4.1819</pub-id>, PMID: <pub-id pub-id-type="pmid">11290733</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname> <given-names>R. S.</given-names>
</name>
<name>
<surname>DuVal</surname> <given-names>A. E.</given-names>
</name>
<name>
<surname>Jensen</surname> <given-names>H. R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Patterns and processes in crop domestication: an historical review and quantitative analysis of 203 global food crops</article-title>. <source>New Phytol.</source> <volume>196</volume>, <fpage>29</fpage>&#x2013;<lpage>48</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1469-8137.2012.04253.x</pub-id>, PMID: <pub-id pub-id-type="pmid">22889076</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pal</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Jaiswal</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Chauhan</surname> <given-names>R. S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>DRPPP: A machine learning based tool for prediction of disease resistance proteins in plants</article-title>. <source>Comput. Biol. Med.</source> <volume>78</volume>, <fpage>42</fpage>&#x2013;<lpage>48</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2016.09.008</pub-id>, PMID: <pub-id pub-id-type="pmid">27658260</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hou</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Identifying nucleotide-binding leucine-rich repeat receptor and pathogen effector pairing using transfer-learning and bilinear attention network</article-title>. <source>Bioinformatics</source> <volume>40</volume>, <elocation-id>btae581</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btae581</pub-id>, PMID: <pub-id pub-id-type="pmid">39331576</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Dan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Genomic insights into local adaptation and future climate-induced vulnerability of a keystone forest tree in East Asia</article-title>. <source>Nat. Commun.</source> <volume>13</volume>, <fpage>6541</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-022-34206-8</pub-id>, PMID: <pub-id pub-id-type="pmid">36319648</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Akter</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Rogers</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Grene</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Prediction of condition-specific regulatory genes using machine learning</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume>, <fpage>e62</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkaa264</pub-id>, PMID: <pub-id pub-id-type="pmid">32329779</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Genetic modification of G&#x3b3; subunit AT1 enhances salt-alkali tolerance in main graminaceous crops</article-title>. <source>Natl. Sci. Rev.</source> <volume>10</volume>, <elocation-id>nwad075</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nsr/nwad075</pub-id>, PMID: <pub-id pub-id-type="pmid">37181090</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tanha</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Abdi</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Samadi</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Razzaghi</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Asadpour</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Boosting methods for multi-class imbalanced data classification: an experimental review</article-title>. <source>J. Big Data</source> <volume>7</volume>, <fpage>70</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s40537-020-00349-y</pub-id>
</citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Dijk</surname> <given-names>A. D. J.</given-names>
</name>
<name>
<surname>Kootstra</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Kruijer</surname> <given-names>W.</given-names>
</name>
<name>
<surname>de Ridder</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Machine learning in plant science and plant breeding</article-title>. <source>iScience</source> <volume>24</volume>, <elocation-id>101890</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.isci.2020.101890</pub-id>, PMID: <pub-id pub-id-type="pmid">33364579</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wallace</surname> <given-names>J. G.</given-names>
</name>
<name>
<surname>Rodgers-Melnick</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Buckler</surname> <given-names>E. S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>On the road to breeding 4.0: unraveling the good, the bad, and the boring of crop quantitative genomics</article-title>. <source>Annu. Rev. Genet.</source> <volume>52</volume>, <fpage>421</fpage>&#x2013;<lpage>444</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev-genet-120116-024846</pub-id>, PMID: <pub-id pub-id-type="pmid">30285496</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>PGD: a machine learning-based photosynthetic-related gene detection approach</article-title>. <source>BMC Bioinf.</source> <volume>23</volume>, <fpage>183</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-022-04722-x</pub-id>, PMID: <pub-id pub-id-type="pmid">35581553</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ferjani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Genetic variation&#xa0;in ZmVPP1 contributes to drought tolerance in maize seedlings</article-title>. <source>Nat.&#xa0;Genet.</source> <volume>48</volume>, <fpage>1233</fpage>&#x2013;<lpage>1241</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ng.3636</pub-id>, PMID: <pub-id pub-id-type="pmid">27526320</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wray</surname> <given-names>N. R.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hayes</surname> <given-names>B. J.</given-names>
</name>
<name>
<surname>Price</surname> <given-names>A. L.</given-names>
</name>
<name>
<surname>Goddard</surname> <given-names>M. E.</given-names>
</name>
<name>
<surname>Visscher</surname> <given-names>P. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Pitfalls of predicting complex traits from SNPs</article-title>. <source>Nat. Rev. Genet.</source> <volume>14</volume>, <fpage>507</fpage>&#x2013;<lpage>515</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrg3457</pub-id>, PMID: <pub-id pub-id-type="pmid">23774735</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gui</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Deep learning in template-free <italic>de novo</italic> biosynthetic pathway design of natural products</article-title>. <source>Brief Bioinform.</source> <volume>25</volume>, <elocation-id>bbae495</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbae495</pub-id>, PMID: <pub-id pub-id-type="pmid">39373052</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Machine learning bridges omics sciences and plant breeding</article-title>. <source>Trends Plant Sci.</source> <volume>28</volume>, <fpage>199</fpage>&#x2013;<lpage>210</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.tplants.2022.08.018</pub-id>, PMID: <pub-id pub-id-type="pmid">36153276</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Plant phenomics and high-throughput phenotyping: accelerating rice functional genomics using multidisciplinary technologies</article-title>. <source>Curr. Opin. Plant Biol.</source> <volume>16</volume>, <fpage>180</fpage>&#x2013;<lpage>187</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.pbi.2013.03.005</pub-id>, PMID: <pub-id pub-id-type="pmid">23578473</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Xue</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Glyphosate-induced GhAG2 is involved in resistance to salt stress in cotton</article-title>. <source>Plant Cell Rep.</source> <volume>41</volume>, <fpage>1131</fpage>&#x2013;<lpage>1145</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00299-022-02844-3</pub-id>, PMID: <pub-id pub-id-type="pmid">35243542</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zafar</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Razzaq</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chattha</surname> <given-names>W. S.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Parvaiz</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Amin</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Investigation of salt tolerance in cotton germplasm by analyzing agro-physiological traits and ERF genes expression</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>11809</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-024-60778-0</pub-id>, PMID: <pub-id pub-id-type="pmid">38782928</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Su</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>The proline-rich protein MdPRP6 confers tolerance to salt stress in transgenic apple (Malus domestica)</article-title>. <source>Scientia Hortic.</source> <volume>308</volume>, <elocation-id>111581</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scienta.2022.111581</pub-id>
</citation></ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>J. K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Abiotic stress responses in plants</article-title>. <source>Nat. Rev. Genet.</source> <volume>23</volume>, <fpage>104</fpage>&#x2013;<lpage>119</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41576-021-00413-0</pub-id>, PMID: <pub-id pub-id-type="pmid">34561623</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Gui</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Hou</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>GwasWA: A GWAS one-stop analysis platform from WGS data to variant effect assessment</article-title>. <source>Comput. Biol. Med.</source> <volume>169</volume>, <elocation-id>107820</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107820</pub-id>, PMID: <pub-id pub-id-type="pmid">38113679</pub-id></citation></ref>
</ref-list>
</back>
</article>