<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1659345</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Uncovering dormancy stage predictors in sweet cherry through DNA methylation and machine learning integration</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Saavedra</surname>
<given-names>Gabriela M.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2588928/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Povea</surname>
<given-names>Poliana</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3177913/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Urra</surname>
<given-names>Claudio</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1849265/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gaete-Loyola</surname>
<given-names>Jos&#xe9;</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3177849/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Maldonado</surname>
<given-names>Carlos</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/603430/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Almeida</surname>
<given-names>Andrea Miyasaka</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/225145/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Centro de Gen&#xf3;mica y Bioinform&#xe1;tica, Facultad de Ciencias, Ingenier&#xed;a y Tecnolog&#xed;a, Universidad Mayor</institution>, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Programa de Doctorado en Gen&#xf3;mica Integrativa, Vicerrector&#xed;a de Investigaci&#xf3;n, Universidad Mayor</institution>, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Escuela de Agronom&#xed;a, Facultad de Ciencias, Ingenier&#xed;a y Tecnolog&#xed;a, Universidad Mayor</institution>, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/882776/overview">Shanwen Sun</ext-link>, Northeast Forestry University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1835978/overview">Jun Yan</ext-link>, China Agricultural University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3126521/overview">Zihui Zhang</ext-link>, Northeast Forestry University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Carlos Maldonado, <email xlink:href="mailto:carlos.maldonado@umayor.cl">carlos.maldonado@umayor.cl</email>; Andrea Miyasaka Almeida, <email xlink:href="mailto:andrea.miyasaka@umayor.cl">andrea.miyasaka@umayor.cl</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1659345</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Saavedra, Povea, Urra, Gaete-Loyola, Maldonado and Almeida.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Saavedra, Povea, Urra, Gaete-Loyola, Maldonado and Almeida</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>
<italic>Prunus Avium</italic> L. dormancy is a complex physiological process that allows floral outbreaks to survive adverse winter conditions and resume favorable spring growth. Traditional phenological evaluations and agroclimatic models, although widely used, exhibit limited resolution and robustness over the years and cultivars. Epigenetic mechanisms, particularly DNA methylation, have emerged as critical regulators of dormancy transitions. However, the integration of methylation data with automatic learning tools (ML) for predictive modeling remains largely unexplored in perennial species. This study presents an integrative frame that combines whole-genome bisulfite sequencing and supervised ML to identify methylation markers at the cytosine and region level associated with specific dormancy stages in the sweet cherry.</p>
</sec>
<sec>
<title>Methods</title>
<p>DNA methylation data sets from three different experiments underwent classification using Random Forest (RF) and eXtreme Gradient Boosting (XGBoost), complemented by SHapley Additive exPlanations (SHAP) for interpretability. The importance of the features was evaluated using the Integrated Model consensus in the RF, XGBoost, and SHAP metrics.</p>
</sec>
<sec>
<title>Results</title>
<p>The selection of features significantly improved the classification performance in the three-stages models (paradormancy, endodormancy, ecodormancy) and two-stages (endodormancy and ecodormancy). RF constantly exceeded XGBoost, achieving an accuracy of up to 97.1% in the two-stages scenario using informative cytosine level data. The SHAP analyses demonstrated that the selected feature effectively discriminated among stages of dormancy and revealed biologically significant epigenetic features. The key features were distributed not random throughout the genome, often colocalizing with transposable elements of long terminal repetition (LTR), particularly LTR/ty3-retrotransposons and LTR/copia families. Some features also co-localize with QTLs for chilling and heat requirement, flowering time and maturity date previously identified.</p>
</sec>
<sec>
<title>Conclusions</title>
<p>This study highlights the usefulness of combining high-resolution methylation data with interpretable ML techniques to identify robust dormancy biomarkers. The enrichment of the features associated with dormancy within the transposable elements and the proximal regions of genes suggests an epigenetic regulation through the remodeling of chromatin mediated by TE. These findings contribute to a deeper understanding of dormancy mechanisms and offer a basis for the development of non-destructive tools based on methylation to improve phenological management in perennial fruit crops.</p>
</sec>
</abstract>
<kwd-group>
<kwd>stage predictive model</kwd>
<kwd>epigenetics</kwd>
<kwd>biomarkers</kwd>
<kwd>
<italic>Prunus avium</italic>
</kwd>
<kwd>feature selection</kwd>
<kwd>bud break</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="1"/>
<equation-count count="3"/>
<ref-count count="55"/>
<page-count count="17"/>
<word-count count="9429"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Functional and Applied Plant Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Dormancy in fruit species such as cherry (<italic>Prunus avium</italic> L.) is an essential process that synchronizes flower bud growth and reproduction with seasonal cycles (<xref ref-type="bibr" rid="B16">Considine and Considine, 2016</xref>). The bud dormancy has been classified into three widely accepted phases: paradormancy, endodormancy, and ecodormancy, which respond to physiological and environmental signals, allowing buds to survive adverse winter conditions and resume development in spring (<xref ref-type="bibr" rid="B15">Cline and Deppong, 1999</xref>; <xref ref-type="bibr" rid="B16">Considine and Considine, 2016</xref>). The dormancy has generally been assessed using destructive phenological assays (e.g., forcing tests; <xref ref-type="bibr" rid="B4">Baumgarten et&#xa0;al., 2021</xref>) and agroclimatic models (chill hours, chill units, or chill portions; <xref ref-type="bibr" rid="B34">Luedeling et&#xa0;al., 2011</xref>; <xref ref-type="bibr" rid="B4">Baumgarten et&#xa0;al., 2021</xref>). However, these approaches have significant limitations in terms of accuracy and interannual application due to dynamically changing environmental conditions (<xref ref-type="bibr" rid="B2">Alonso-Gonz&#xe1;lez et&#xa0;al., 2020</xref>). Therefore, several studies have sought molecular biomarkers that allow more accurate, rapid, and noninvasive monitoring of bud physiological status.</p>
<p>During the past decade, progress has been made in the study of the mechanisms associated with dormancy stages in various species, improving the general understanding of dormancy transitions in different species (<xref ref-type="bibr" rid="B43">Rothkegel et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B50">Yang et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B44">Rothkegel et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B17">Ding et&#xa0;al., 2024</xref>). These studies have revealed that dormancy phases are regulated by DORMANCY-ASSOCIATED MADS-box (DAM) genes, phytohormones, carbohydrates, temperature, photoperiod, reactive oxygen species, water deprivation, cold acclimation, and epigenetic regulation. Particularly, DNA methylation has emerged as one of the key epigenetic mechanisms involved in plant development and environmental responses. In this sense, dynamic changes in genome methylation patterns throughout the dormancy cycle, suggesting a functional role for this epigenetic mark in determining bud fate (<xref ref-type="bibr" rid="B44">Rothkegel et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B50">Yang et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B30">Kumar et&#xa0;al., 2016</xref>). In this sense, <xref ref-type="bibr" rid="B43">Rothkegel et&#xa0;al. (2017)</xref> indicated that in <italic>Prunus avium</italic> L, genome methylation patterns are one of the mechanisms that regulate the MADS-box genes controlling bud dormancy. Similarly, <xref ref-type="bibr" rid="B30">Kumar et&#xa0;al. (2016)</xref> showed that DNA methylation has been linked to chilling acquisition during dormancy in <italic>Malus domestica</italic>. Additionally, high levels of DNA methylation have been observed during the induction of endodormant floral buds in blueberry compared to those in the ecodormant stage (<xref ref-type="bibr" rid="B31">Li et&#xa0;al., 2015</xref>), which indicates a dynamic regulation throughout the dormancy stages.</p>
<p>The development of high-throughput sequencing technologies, such as whole-genome bisulfite sequencing (WGBS), enables the precise examination of methylation status at individual CpG sites with high resolution (<xref ref-type="bibr" rid="B54">Zhou et&#xa0;al., 2019</xref>). However, this approach generates highly complex and high-dimensional datasets, where the number of features (methylated sites) far exceeds the number of available samples. This problem, coupled with biological variability between cultivars and climatic heterogeneity between seasons, poses significant challenges for conventional statistical analyses and hampers biomarker discovery and straightforward biological interpretation (<xref ref-type="bibr" rid="B53">Zhang et&#xa0;al., 2018</xref>).</p>
<p>Artificial intelligence (AI)-based approaches have shown great promise for the analysis of complex and high-dimensional molecular data (<xref ref-type="bibr" rid="B8">Bzdok et&#xa0;al., 2018</xref>). However, one of the main limitations of these algorithms in biology is their interpretability and identification of biologically relevant variables. In this sense, feature selection is essential to eliminate noisy features, improve model performance, and optimize interpretability (<xref ref-type="bibr" rid="B35">Lundberg and Lee, 2017</xref>). Ensemble learning algorithms, such as Random Forest&#xa0;(RF) and eXtreme Gradient Boosting (XGBoost), have demonstrated exceptional performance in genomic studies due to their ability to handle noisy, imbalanced, and high-dimensional data while providing internal feature importance metrics (<xref ref-type="bibr" rid="B42">Raihan&#xa0;et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B39">Peng and Yu, 2024</xref>; <xref ref-type="bibr" rid="B14">Chu et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B32">Li et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B52">Zhang et&#xa0;al., 2024</xref>). Furthermore, the SHapley Additive exPlanations (SHAP) method offers a game-theoretic interpretation of each variable&#x2019;s individual contribution to model predictions (<xref ref-type="bibr" rid="B35">Lundberg and Lee, 2017</xref>). These models (RF, XGB, and SHAP) have been successfully used to identify important variables in predictive models for human diseases (<xref ref-type="bibr" rid="B42">Raihan et&#xa0;al., 2023</xref>), agronomic traits (<xref ref-type="bibr" rid="B52">Zhang et&#xa0;al., 2024</xref>), precision agriculture (<xref ref-type="bibr" rid="B32">Li et&#xa0;al., 2024</xref>), and seed viability (<xref ref-type="bibr" rid="B14">Chu et&#xa0;al., 2024</xref>).</p>
<p>In this study, DNA methylation profiles obtained via WGBS&#xa0;with&#xa0;supervised Machine Learning (ML) algorithms (RF and XGB) and SHAP interpretability analyses were integrated to identify informative cytosines and methylated regions capable of discriminating between different dormancy stages in sweet cherry floral buds. Classification models were constructed for two scenarios: (i) three classes (paradormancy, endodormancy, and ecodormancy stages) and (ii) two classes (endodormancy and ecodormancy stages). This integrative strategy not only improves classification accuracy of dormancy stages but also provides deeper insights into the epigenetic mechanisms governing this process, offering potential biomarkers for breeding programs and more precise tools for phenological management in perennial fruit species.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Plant material, sampling, and chilling requirement determination under forcing conditions</title>
<p>Adult (9&#x2013;12-year-old) sweet cherry (<italic>Prunus avium</italic> L.) trees of cultivars Santina and Regina were sampled from two commercial orchards, &#x201c;Morza&#x201d; and &#x201c;Entre R&#xed;os&#x201d;, located in the Maule and O&#x2019;Higgins regions of Chile, respectively during autumn/winter of 2022 (Experiment 3, <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). As experiments were performed in growth chambers (Experiments 1 and 2, <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>) the cold accumulation as chilling hours was calculated according to <xref ref-type="bibr" rid="B48">Weinberger (1950)</xref>. To facilitate more comprehensible data comparison labels, all chilling accumulation measurements were transformed into chilling hours. A second experiment using Regina cultivar was also performed. Adult (8&#x2013;10-year-old) sweet cherry &#x2018;Regina&#x2019; trees, which were part of the INIA Sweet Cherry Breeding Program collection located at the Los Tilos Station, Metropolitan Region of Chile were used. Samples of corresponding branches bearing floral buds were obtained from this orchard at the beginning of autumn 2021 (April in the Southern hemisphere) and subjected to continuous chilling as described by <xref ref-type="bibr" rid="B46">Soto et&#xa0;al. (2022)</xref>. Briefly, the collected branches were transported to the laboratory, disinfected, separated into lots of 4&#x2013;5 branches, and stored at 4&#x2013;6 &#xb0;C for progressive chilling accumulation.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Summary of evaluated cultivars, trial locations, required chilling hours per cultivar, chilling hours sampling, and year of the trial.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Cultivar</th>
<th valign="middle" align="center">Fields</th>
<th valign="middle" align="center">Chill requirement</th>
<th valign="middle" align="center">Chill Hour (CH) sampling</th>
<th valign="middle" align="center">Year</th>
<th valign="middle" align="center">Experiment</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Royal Dawn (RD)</td>
<td valign="middle" align="center">Mostazal</td>
<td valign="middle" align="center">350&#x2013;500 CH</td>
<td valign="middle" align="center">0 &#x2013; 173 &#x2013; 348 &#x2013; 516</td>
<td valign="middle" align="center">2015</td>
<td valign="middle" align="center">Experiment 1</td>
</tr>
<tr>
<td valign="middle" align="center">Kordia (K)</td>
<td valign="middle" align="center">Quillota</td>
<td valign="middle" align="center">1450&#x2013;1600 CH</td>
<td valign="middle" align="center">0 &#x2013; 443 &#x2013; 1295 - 1637</td>
<td valign="middle" align="center">2015</td>
<td valign="middle" align="center">Experiment 1</td>
</tr>
<tr>
<td valign="middle" align="center">Santina (S)</td>
<td valign="middle" align="center">Morza</td>
<td valign="middle" align="center">600&#x2013;800 CH</td>
<td valign="middle" align="center">269 - 599 - 973 - 1234</td>
<td valign="middle" align="center">2022</td>
<td valign="middle" align="center">Experiment 3</td>
</tr>
<tr>
<td valign="middle" rowspan="3" align="center">Regina (R)</td>
<td valign="middle" align="center">Los Tilos</td>
<td valign="middle" align="center">1150-1600</td>
<td valign="middle" align="center">200 &#x2013; 1160 - 1700</td>
<td valign="middle" align="center">2021</td>
<td valign="middle" align="center">Experiment 2</td>
</tr>
<tr>
<td valign="middle" align="center">Entre R&#xed;os (E)</td>
<td valign="middle" align="center">770&#x2013;830 CH</td>
<td valign="middle" align="center">202 - 463 - 769 - 837 - 959</td>
<td valign="middle" align="center">2022</td>
<td valign="middle" align="center">Experiment 3</td>
</tr>
<tr>
<td valign="middle" align="center">Morza (M)</td>
<td valign="middle" align="center">600&#x2013;800 CH</td>
<td valign="middle" align="center">268 - 599 - 973 - 1234</td>
<td valign="middle" align="center">2022</td>
<td valign="middle" align="center">Experiment 3</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Floral buds were collected from each cultivar at different chill accumulation time points, frozen in liquid nitrogen, and transferred to our laboratory facilities. The dormancy status at each time point was assessed through forcing experiments. Bud break percentages were recorded during 14 days at 25&#xb0;C under a 16/8 h day/night photoperiod. After 14 days, the phenological status of floral buds was scored, and the chilling requirement was considered fulfilled when at least 50% of the buds burst (BBCH 51 stage; <xref ref-type="bibr" rid="B19">Fad&#xf3;n et&#xa0;al., 2015</xref>).</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Yield and quality analysis of isolated DNA and sequencing</title>
<p>DNA was extracted from 100 mg of ground frozen tissue using the DNeasy Plant Mini Kit (Qiagen) following the manufacturer&#x2019;s procedure. The DNA eluted in 50 &#xb5;l of water was quantified by Qubit DNA High Sensitivity fluorometry assay (Life Technologies). Its integrity was evaluated on a 0.8% agarose gel. For library preparation, a 150 bp paired-end sequencing strategy was conducted by Novogene (USA). Library preparation and sequencing were performed in an Illumina Novaseq System to generate ~ 20 Gbp of data per sample.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>BS-seq datasets and processing</title>
<p>Kordia and Royal Dawn cultivars&#x2019; raw data were retrieved from the SRA database (PRJNA610988 and PRJNA610989). These datasets collection and processing procedures were described by <xref ref-type="bibr" rid="B44">Rothkegel et&#xa0;al. (2020)</xref>. Raw data quality analysis was performed on samples of Kordia (K_2015), Royal Dawn (RD_2015), Regina Los Tilos (R_2021), Regina Morza (R_M_2022), Regina Entre R&#xed;os (R_E_2022) and Santina Morza (S_M_2022). Quality assessments were performed with FastQC v0.12.1 (<xref ref-type="bibr" rid="B3">Andrews, 2010</xref>) software and trimmed with Trim Galore (<xref ref-type="bibr" rid="B28">Krueger et&#xa0;al., 2016</xref>), which cuts adaptors and improves raw data quality (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>). Clean bisulfite-treated data were then analyzed with the Bismark software (<xref ref-type="bibr" rid="B27">Krueger and Andrews, 2011</xref>), then mapped to <italic>Prunus avium</italic> Tieton v2.0 assembly as the reference genome (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>). A deduplication step was required to remove duplicated alignments caused by excessive PCR amplification. Then, we proceed with the methylation extraction step to identify positions of every cytosine in the corresponding DNA methylation context: CG, CHG, and CHH (where H is A, C, or T) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>). Datasets representing methylated cytosines and methylated genomic regions were generated following the BS-seq Analysis Workflow (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Workflows. Bisulfite sequencing analysis <bold>(A)</bold> and Machine Learning-based selection of informative features <bold>(B)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1659345-g001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a genomic analysis workflow in two sections. A) Steps include whole-genome bisulfite sequencing, quality control, mapping to the Prunus avium genome, deduplication, and methylation extraction, leading to cytosine coverage reports and final regions. B) Selection of features using classification trees and SHAP values for machine learning. A table lists cytosine features with scores from different methods. Diagrams depict identification of genomic regions and improved classification of dormancy stages.</alt-text>
</graphic>
</fig>
<p>CpG coverage was used to generate a cytosine database used as an ML algorithm input. In this database, filters were applied. For individual cytosines, consider at least 4 reads of coverage by cytosine and present among all samples, resulting in 716,255 cytosines. The filter applied by region was in the previously filtered cytosines, 4 or more cytosines were looked for within bins of 100 bp, with a gap of 100 bp between bins. This filter resulted in 69,398 regions. This filtered data was used as input for the classification analysis, using both RF and XGBoost as machine learning algorithms.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Classification and interpretation methodologies for variables</title>
<p>Two classification algorithms (RF and XGBoost) were employed to model the classification task. In addition, SHAP values were used to interpret the classification models, identifying the most informative features contributing to predictions (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>). The SHAP-derived feature importance scores were compared against the built-in feature importance metrics of both XGBoost and RF enabling a comprehensive evaluation of key features relevant to the classification task.</p>
<sec id="s2_4_1">
<label>2.4.1</label>
<title>Random forest</title>
<p>The RF algorithm was employed in this study due to its robustness, interpretability, and its capacity to process high-dimensional datasets that may contain irrelevant or noisy features. RF is an ensemble learning method that constructs a collection of decision trees using two key sources of randomness: (i) bootstrap sampling, which involves drawing random subsets of the training data with replacement, and (ii) random feature selection, where only a randomly chosen subset of features is considered at each decision node when splitting. This dual randomization strategy helps to reduce model variance and overfitting, improving generalization (<xref ref-type="bibr" rid="B33">Liu et&#xa0;al., 2012</xref>). For training, the original dataset is divided into in-bag samples (used to build individual trees) and out-of-bag samples, which serve as an internal validation set to assess model performance. Approximately two-thirds of the data are used for training, while the remaining third is reserved for out-of-bag error estimation, which provides an unbiased measure of predictive accuracy (<xref ref-type="bibr" rid="B26">Kavzoglu and Teke, 2022</xref>). Each decision tree in the ensemble outputs a class prediction, and the RF model aggregates these results to determine the final classification. This aggregation is done using a majority voting mechanism, formally defined as (<xref ref-type="bibr" rid="B7">Breiman, 1996</xref>):</p>
<disp-formula>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:munderover>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>H(x)</italic> is the final prediction of the RF model for instance <italic>x</italic>, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the prediction of the <italic>i</italic>-th decision tree, <italic>I</italic> is the indicator function, and <italic>Y</italic> represents a class label. This voting scheme ensures that the final decision reflects the consensus of the ensemble. In addition to its predictive capabilities, RF also provides an intrinsic measure of feature importance. The Gini index measures the decrease in node impurity contributed by each variable. A higher Gini importance score indicates a greater contribution to the classification process.</p>
</sec>
<sec id="s2_4_2">
<label>2.4.2</label>
<title>XGBoost</title>
<p>The XGBoost algorithm was utilized in this study as a complementary classification method due to its high predictive performance and ability to handle noisy, high-dimensional biological data (<xref ref-type="bibr" rid="B26">Kavzoglu and Teke, 2022</xref>). Originally proposed by <xref ref-type="bibr" rid="B13">Chen and Guestrin (2016)</xref>, XGBoost is based on the gradient boosting framework and builds models sequentially, where each subsequent tree attempts to correct the residual errors of the ensemble constructed so far. Unlike Random Forest, which builds trees independently, XGBoost optimizes performance by minimizing a regularized objective function in a stage-wise manner.</p>
<p>At each iteration t, the algorithm adds a new function ft to improve the model&#x2019;s prediction. The overall objective function can be expressed as (<xref ref-type="bibr" rid="B13">Chen and Guestrin, 2016</xref>):</p>
<p>
<inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2205;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
<mml:mo>+</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>k</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mi>&#x3a9;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<disp-formula>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtext>with&#xa0;&#x3a9;</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>w</mml:mi>
<mml:msup>
<mml:mo>|</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>l</italic> is a differentiable convex loss function that measures the difference between the prediction <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the target <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The regularization term <italic>&#x2126;</italic> penalizes the complexity of the model, thereby controlling model complexity and reducing the risk of overfitting. For this, <italic>T</italic> is the number of leaf nodes in the tree, w is the score of each leaf, <inline-formula>
<mml:math display="inline" id="im5">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im6">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> are regularization parameters that control the penalty for model complexity. This formulation encourages the model to produce trees with fewer and more informative splits, improving generalization. XGBoost also implements strategies such as learning rate and column subsampling to further prevent overfitting.</p>
</sec>
<sec id="s2_4_3">
<label>2.4.3</label>
<title>SHAP</title>
<p>To enhance model interpretability and ensure transparency in the classification process, this study applied SHAP as a <italic>post hoc</italic> interpretability method. SHAP, introduced by <xref ref-type="bibr" rid="B35">Lundberg and Lee (2017)</xref>, is a unified framework rooted in cooperative game theory that quantifies the contribution of each feature to a model&#x2019;s prediction. Unlike traditional feature importance scores generated by ensemble algorithms, SHAP values allow for both global and local interpretability, revealing not only which features are important but also how they positively or negatively influence specific predictions.</p>
<p>The SHAP methodology assigns an importance value &#x3c6;<italic>
<sub>i</sub>
</italic> to each feature, representing the marginal contribution of including that feature in the prediction model, averaged over all possible feature subsets. Formally, the Shapley value is defined as:</p>
<disp-formula>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c6;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2286;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>!</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>!</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>!</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>F</italic> is the full set of input features, <italic>S</italic> is a subset of features not containing <italic>i</italic>, <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are retrained, and predictions of these two models are compared to the current input <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (<inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>)- <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the values of the input features in the set <italic>S</italic>. This formulation ensures both local accuracy and consistency, two desirable properties for interpretable machine learning (<xref ref-type="bibr" rid="B5">Bi et&#xa0;al., 2020</xref>).</p>
</sec>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Feature selection via integrated model (RF, XGB, and SHAP)</title>
<p>To identify the most relevant and biologically informative features, a comprehensive feature selection strategy was implemented by integrating importance metrics from four distinct perspectives: RF, XGB, and SHAP computed over both RF and XGB models. Each method was used to independently estimate the relevance of input features based on either intrinsic model properties or <italic>post hoc</italic> explanations.</p>
<p>Feature importance scores were derived directly from the trained RF and XGB models using their respective built-in ranking mechanisms. In parallel, SHAP values were calculated to generate explanations for the predictions of both the RF and XGB models.</p>
<p>To ensure robustness and consistency, only features that exhibited non-zero importance across all four models: RF, XGB, SHAP(RF), and SHAP(XGB) were retained. This intersection-based filtering process served three main purposes: (i) to reduce the number of input features fed into the final models, (ii) to identify components with consistent predictive relevance and minimal redundancy, and (iii) to lower computational complexity and improve model generalizability. Additionally, by retaining only consistently informative methylation features, the models achieved improved interpretability while maintaining high classification performance in dormancy stage prediction.</p>
<p>All computations were carried out using Python (v3.12.7). RF and XGB models were implemented using the sklearn and xgboost libraries, while SHAP analyses were conducted via the SHAP library. All codes are provided in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Code S1</bold>
</xref>.</p>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Model evaluation and performance metrics</title>
<p>A comprehensive set of evaluation metrics was employed, including accuracy, precision, recall and F1-score. These metrics provide a balanced view of classifier performance, particularly in datasets with class imbalance, as they capture both sensitivity and specificity of the predictions. The confusion matrix was also utilized as an intuitive and informative tool for summarizing classification outcomes across true positives, true negatives, false positives, and false negatives. Additionally, the Receiver Operating Characteristic (ROC) curve was used to visualize the trade-off between the true positive rate and the false positive rate at various decision thresholds. The area under the ROC curve metric, derived from this curve, was used to quantify the model&#x2019;s ability to distinguish between classes.</p>
<p>All metrics were averaged over multiple runs using Repeated Stratified k-Fold Cross-Validation. This method maintains the class distribution within each fold and reduces variance due to random sampling. Two different k-fold strategies were applied depending on the class distribution:</p>
<p>i) For the classification of paradormancy, ecodormancy and endodormancy, where the minority class had only 9 instances (paradormancy), a 3-fold repeated stratified cross-validation was used to ensure that each fold contained at least one instance from each class.</p>
<p>ii) For the classification of endormancy and ecodormancy, where the smallest class had 26 samples (ecodormancy), a 10-fold repeated stratified cross-validation was performed.</p>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>DNA methylation quantification, feature annotation, and visualization</title>
<p>WGBS data was processed to quantify cytosine methylation levels. Alignment metrics were parsed in R to calculate absolute and relative methylation levels across CpG, CHG, and CHH contexts. Absolute methylation percentages were computed per sample based on the ratio of methylated to total cytosines. For relative methylation, only methylated cytosines were considered, and their context-specific contributions were normalized to derive proportions, which were visualized per sample and chilling hour accumulation. Relevant cytosines were further classified by dormancy model (2-stages and 3-stages) and feature type (cytosine or region) and formatted into BED. Feature widths were calculated and their distributions compared globally and per model using histograms. Then we looked for shared and unique features across models. Features selected from the models were then classified into their genomic contexts: promoter (defined as 2kb upstream from the TSS), gene bodies, downstream region (2kb downstream from the transcription termination site), intergenic region (all locations different from promoter, 5&#x2032;-UTR, exon, intron, 3&#x2032;-UTR, or downstream), and a transposable elements (TEs) annotation developed by our group.</p>
</sec>
<sec id="s2_8">
<label>2.8</label>
<title>
<italic>P. avium</italic> Tieton TEs annotation and TE-features downstream analysis</title>
<p>In order to identify TEs in <italic>P. avium</italic> Tieton, a <italic>de novo</italic> annotation using the RepeatModeler v2.0.5 tool was performed (<xref ref-type="bibr" rid="B21">Flynn et&#xa0;al., 2020</xref>). Since we lacked a reliable and curated database to search for specific transposable elements in <italic>P. avium</italic>. Using the BuildDatabase module, an index to use as RepeatModeler input was generated. This process writes a classification file that is then used as input for RepeatMasker v4.1.5 (<xref ref-type="bibr" rid="B45">Smit et&#xa0;al., 2019</xref>) and outputs a detailed annotation of identified transposable elements. Then the distribution of these TEs was characterized, since many features colocalized with TEs. Package BEDTools was used to intersect features from the classification models with genomic contexts: promoters, gene bodies, downstream region, intergenic region, and the TE annotation previously developed by our group. TE-associated to model relevant features were summarized by class and genomic context and visualized to identify context-specific enrichment patterns. All analyses and visualizations were performed in R. A circo plot showing this was made using the R packages: circlize, GenomicRanges, rtracklayer, and Biostrings; <italic>P. avium</italic> Tieton annotation file, and our TE annotation. Then, filtered by long terminal repeats (LTR) and splitted by chromosomes in 100 kb windows were added tracks of: LTR/ty3-retrotransposons, LTR/Copia, and the sum of both.</p>
</sec>
<sec id="s2_9">
<label>2.9</label>
<title>Features colocalization with quantitative trait loci</title>
<p>The distribution of methylated cytosines and regions on chromosome 4 was analyzed since several studies have identified QTLs associated with dormancy-related traits in sweet cherry. For this, four QTLs that were previously developed and associated with chilling requirements (CR), heat requirements (HR), flowering date (FD) (<xref ref-type="bibr" rid="B12">Cast&#xe8;de et&#xa0;al., 2014</xref>), and fruit maturity date (MD) (<xref ref-type="bibr" rid="B10">Calle and W&#xfc;nsch, 2020</xref>) were used.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Whole-genome bisulfite sequencing of cultivars of <italic>P. avium</italic>
</title>
<p>By understanding variations in DNA methylation and the context where methylated cytosine belongs, the regulatory roles of methylation in the genome can be discovered. The pattern and abundance of each mC-context contribute to the regulation of gene expression, transposon repression, or adaptation to environmental changes, among other functions in plants. The relative contribution of the three main DNA methylation contexts in sweet cherry cultivars and sampling years (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>) shows that across all samples, mCG consistently represents the largest proportion of methylated cytosines, with values ranging from 40% to 51%. Followed by the mCHG context, whose proportions are typically between 26% and 34%, while mCHH constitutes the smallest fraction, with values ranging from 21% to 32%. Overall, while the relative contributions of each methylation context are broadly similar across cultivars and years, subtle differences are evident among samples and conditions. Notably, samples from Experiment 3 Regina (Entre R&#xed;os, 2022) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>) present the highest mCG and lowest mCHH proportions. Across all cultivars, mCHG values remain intermediate and relatively stable. Additionally, absolute methylation levels were computed in each sample (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>). CpG context is highly methylated in most of the samples. Experiment 1 (Kordia and Royal Dawn, 2015) showed a particularly high mCHG context, in both cultivars subject of study.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Feature subset selection</title>
<p>Feature selection was conducted independently for each input data type (cytosine-associated methylation and region-specific methylation) and under two classification scenarios: one including all three dormancy stages (3-stages: paradormancy, endodormancy, ecodormancy), and another using only endodormancy and ecodormancy samples (2-stages). In the three-class scenario, the cytosine dataset was reduced from 716,255 to 64 features, and the region-based dataset from 69,398 to 217 features. For the two-class setting, dimensionality was reduced to 297 and 535 features for the cytosine and regional datasets, respectively.</p>
<p>To assess the impact of feature selection on the underlying data structure, we conducted a t-SNE analysis using both the full and the reduced feature sets. The initial analysis with all features primarily revealed sample clustering based on varietal identity across all comparisons, including both three-stage and two-stage (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>) analyses of cytosine and regional methylation profiles. In contrast, when the analysis was limited to the subset of selected informative features, the resulting t-SNE plots demonstrated a clearer separation of samples according to dormancy stages (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>t-SNE visualization of cytosine- and region-level methylation profiles before and after feature selection across different model configurations. <bold>(A, B)</bold> Visualizations using the full feature sets in the three-stage model: cytosines <bold>(A)</bold> and methylated regions <bold>(B)</bold>. <bold>(C, D)</bold> Visualizations using only the informative features in the three-stage model: cytosines <bold>(C)</bold> and methylated regions <bold>(D)</bold>. <bold>(E, F)</bold> Visualizations using the full feature sets in the two-stage model: cytosines <bold>(E)</bold> and methylated regions <bold>(F)</bold>. <bold>(G, H)</bold> Visualizations using only the informative features in the two-stage model: cytosines <bold>(G)</bold> and methylated regions <bold>(H)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1659345-g002.tif">
<alt-text content-type="machine-generated">Eight scatter plots labeled A to H show t-SNE analysis of different cherry cultivars across various dormancy stages. Each plot differentiates cultivars (Royal Dawn, Kordia, Santina, Regina) with shapes and stages (Paradormancy, Endodormancy, Ecodormancy) with colors. Each graph shows a distinct distribution pattern of points along the t-SNE1 and t-SNE2 axes.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Classification results</title>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Three dormancy stages: paradormancy, endodormancy and ecodormancy</title>
<p>Four distinct methylation datasets were analyzed: two cytosine-level datasets (all features: 716,255; informative features: 64) and two region-level datasets (all features: 69,398; informative features: 217). All datasets were evaluated separately using RF and XGB classifiers to classify samples into three dormancy stages: Ecodormancy, Endodormancy and Paradormancy.</p>
<p>For the cytosine complete (unfiltered) dataset, RF achieved 62.1% accuracy, while XGB reached 59.0%. Similarly, with the full region dataset, RF attained 56.5% accuracy versus XGB&#x2019;s 49.3% (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>). Subsequent analysis using feature-selected datasets revealed significant improvements in classification performance. RF consistently outperformed XGB across all metrics (accuracy, precision, recall, F1-score). With the filtered cytosine dataset (64 informative features), RF achieved 87.3% mean accuracy compared to XGB&#x2019;s 79.1%. The pattern held for the region dataset (217 features), where RF reached 79.1% accuracy versus XGB&#x2019;s 66.4% (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>). The t-tests confirmed that the performance differences between using all features and selected features were statistically significant (p&lt; 0.05) for both classification models.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Performance comparison of RF and XGB models using full features and the informative feature dataset. <bold>(A, B)</bold> Model performances using cytosine-level data <bold>(A)</bold> and methylation region-level data <bold>(B)</bold> in the three-stage classification approach. <bold>(C, D)</bold> Model performances using cytosine-level data <bold>(C)</bold> and methylation region-level data <bold>(D)</bold> in the two-stage classification approach.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1659345-g003.tif">
<alt-text content-type="machine-generated">Four bar charts labeled A, B, C, and D display evaluation metrics for Random Forest (RF) and XGBoost (XGB) models using all and informative features. Metrics include Accuracy, F1-score, Precision, and Recall, with mean values and error bars. Orange bars for RF with informative features consistently show higher values across metrics.</alt-text>
</graphic>
</fig>
<p>Paradormancy remained the most challenging class to predict in all scenarios. Confusion matrices (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S2</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>S3</bold>
</xref>) revealed that both RF and XGB consistently misclassified this class, especially when using the unfiltered datasets. In contrast, classification performance for endodormancy and ecodormancy improved after feature selection, with precision and recall values increasing for both models. The ROC curve analyses (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4</bold>
</xref>) corroborated these findings. The RF model exhibited higher AUC values than XGB across both datasets with all features (cytosines: 0.73, regions: 0.72 for RF; cytosines: 0.73, regions: 0.70 for XGB) and for informative features (cytosines: 0.96, regions: 0.92 for RF; cytosines: 0.92, regions: 0.84 for XGB), further supporting the superiority of RF and the benefit of dimensionality reduction via feature selection.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Two dormancy stages: endodormancy and ecodormancy</title>
<p>Due to the limited number of samples in the paradormancy class, which negatively impacted classification performance, this category was excluded from subsequent analyses. The focus of this stage was thus narrowed to the binary classification between endodormancy and ecodormancy stages, using both the cytosine-level and region-level methylation datasets.</p>
<p>The classification using the complete datasets exhibited low performance. Specifically, RF achieved an accuracy of 70.7% for cytosines and 67.4% for regions, while XGB reached only 44.3%&#xa0;and&#xa0;59.5%, respectively. Confusion matrices from this analysis (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S5</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>S6</bold>
</xref>) revealed considerable misclassification across both models, and ROC analysis confirmed limited discriminative power in both datasets when no feature selection was applied. Classification performance improved markedly following the dimensionality reduction. As shown in (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3C, D</bold>
</xref>), RF achieved a mean accuracy of 97.1% for cytosines and 89.3% for regions, significantly outperforming XGB, which reached only 65.8% and 66.0%, respectively. Evaluation metrics (precision, recall, F1-score) mirrored this trend, and ROC analysis further demonstrated the superior discriminative ability of RF, which attained an AUC of 1.00 in the cytosine dataset and 0.99 in the region-based dataset, compared to 0.68 and 0.67 for XGB (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S7</bold>
</xref>). Moreover, the confusion matrices (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S5</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>S6</bold>
</xref>) illustrate these improvements, in the cytosine dataset, RF misclassified only three ecodormancy samples, whereas XGB misclassified 22 (7 of endodormancy and 15 of ecodormancy). For the regional data, RF made six errors, while XGB misclassified 21 samples in total (8 of endodormancy and 13 of ecodormancy).</p>
<p>These trends were further examined using t-SNE visualizations (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). When applied to the unfiltered datasets, samples tended to cluster primarily according to varietal background rather than dormancy stage. In contrast, the filtered datasets containing the selected informative features revealed more distinct groupings aligned with dormancy stages, although some degree of sample mixing remained. This residual overlap may stem from the inherent limitations of t-SNE, which emphasizes local rather than global structure. However, it is important to note that t-SNE was used solely for visualization purposes and played no role in classification model training.</p>
</sec>
<sec id="s3_3_3">
<label>3.3.3</label>
<title>Explaining the model</title>
<p>Given the superior classification performance of the RF algorithm over XGB, SHAP analysis was applied to the RF models to interpret the contribution of individual features toward dormancy stage classification. This interpretability analysis focused exclusively on the informative feature sets for both the cytosine-level and region-level methylation datasets, as previously described.</p>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A, B</bold>
</xref> illustrates the SHAP summary plots corresponding to the three dormancy stages across both datasets. These visualizations aggregate SHAP values across all samples, providing a ranked overview of feature importance based on their average impact on the model&#x2019;s predictions. In the cytosine dataset (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>), the features chr_4_31092165 (cytosine 31092165 located on chromosome 4) and chr_1_38152240 showed the highest overall impact, particularly in distinguishing the ecodormancy stage, as indicated by their strong SHAP contributions. Additional features such as chr_6_8406189, chr_7_16848224, and chr_4_25227466 also played relevant roles, with varying contributions across the three stages. Interestingly, certain features like chr_4_31050901 and chr_3_23416192 displayed more influence in differentiating paradormancy, suggesting stage-specific relevance. On the other hand, in the regional methylation dataset (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>), top features included chr_3_9331371_9331488 (region situated on chromosome 3, specifically among base pairs 9331371 to 9331488) and chr_1_27137350_27137417, both of which had a particularly strong association with the ecodormancy stage. Several&#xa0;other regions, such as chr_2_7212899_7213096 and chr_7_21605184_21605271, also contributed notably, with more nuanced effects across endodormancy and paradormancy. Notably, some regions, such as chr_6_9727681_9727971 and chr_3_31797583_31797605, exhibited relatively balanced contributions across all three dormancy stages, indicating potential roles in general dormancy regulation. Additionally, <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4C, D</bold>
</xref> also illustrates the comparative feature importance analysis between two dormancy stages (endodormancy and ecodormancy). In the cytosine dataset (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref>), features such as chr_4_31092165 and chr_1_38152240 stood out for their strong influence on the model&#x2019;s predictions, particularly in distinguishing the ecodormancy stage. Other relevant markers included chr_6_8406189 and chr_4_25227466, which showed moderate contributions across both dormancy classes. Meanwhile, the methylation dataset (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4D</bold>
</xref>) highlighted chr_2_7212899_7213096 and chr_3_9331371_9331488 as the most impactful features, especially associated with ecodormancy. Some regions, like chr_6_9727681_9727971, exhibited more balanced SHAP values, suggesting shared roles in dormancy regulation regardless of the specific stage.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>SHAP-based interpretation of feature contributions to dormancy stage classification. <bold>(A, B)</bold> SHAP summary plots showing the top 20 most impactful features for classification, based on <bold>(A)</bold> methylated cytosines and <bold>(B)</bold> methylated regions in the three-stage classification approach. <bold>(C, D)</bold> Global feature importance ranked by mean SHAP values for all samples, highlighting the features with the highest overall contribution using <bold>(C)</bold> cytosine-level and <bold>(D)</bold> region-level methylation data in the two-stage classification approach.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1659345-g004.tif">
<alt-text content-type="machine-generated">Four bar charts labeled A, B, C, and D display the mean SHAP values for various genetic features. Charts A and B include color-coded stages: Paradormancy (orange), Endodormancy (purple), and Ecodormancy (blue), while charts C and D do not have color coding. Each chart shows different features with varying mean SHAP values, indicating the importance of each feature in different contexts.</alt-text>
</graphic>
</fig>
<p>To complement the global analysis, individual SHAP bar plots were generated to explore how specific features contributed to the classification of correctly and incorrectly predicted ecodormancy samples. In <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S8A</bold>
</xref>, the example shows an ecodormancy sample that was correctly classified. Most SHAP values are positive and substantial, particularly for features like chr_1_38152240 and chr_4_31092165, indicating strong support for the ecodormancy prediction. In contrast, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S8B</bold>
</xref> shows an ecodormancy sample that was misclassified as endodormancy by the model. Here, SHAP values for key features either decrease in magnitude or switch direction (e.g., chr_1_38152240 and chr_4_31092165 show negative contributions), effectively shifting the overall model output toward the wrong class. This shift in the SHAP value distribution in this sample highlights how changes in individual feature contributions can drive classification errors.scovery of genomic regions underlying dormancy stages.</p>
<p>To understand the common and unique features of each model, including the overlap of individual cytosines within regions, pairwise and multi-set intersection distribution of the relevant features detected by the two models are shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>. The largest unique subset corresponds to region features from the 2-stages model with 500 regions, followed by 283 unique cytosines. The 3-stages model contributes with 182 unique regions and 58 unique cytosines. Among shared features, only 35 regions and 5 cytosines are common to both models. Relatively few individual cytosines overlap regions, four cytosines in the 2-stages model co-localize with regions from the same model, and another four 2-stages cytosines overlap regions from both models. A single cytosine is shared by the two models and is located within a region feature from the 3-stages model. In addition, no cytosine is common to all four datasets. The 2-stages model yields the largest feature sets and the greatest number of model-unique features.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Features statistics and annotation. <bold>(A)</bold> Venn diagram of relevant features across the two models under study. Colored ellipses display the number of features per set: cytosine 2-stages (red), region 2-stages (light blue), cytosine 3-stages (orange), and region 3-stages (green). Overlaps are also indicated, displaying the number of either cytosines or regions shared, or as unique features per set. <bold>(B)</bold> Genomic annotation of features from the 2-stages model. Cytosine-level and region-level features were detected in the model in five genomic contexts: promoter (2 kb upstream of TSS), gene bodies (exonic and intronic regions), downstream (2 kb downstream 3&#x2019;), transposable element (TE), and intergenic space (regions not overlapping annotated gene-proximal features). <bold>(C)</bold> Class-wise distribution of features from model 2-stages overlapping TEs and their genomic context (Promoter, Gene, and Downstream). Colored bars depicting TE classes based on our curated annotation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1659345-g005.tif">
<alt-text content-type="machine-generated">A) A Venn diagram shows the overlap of cytosine and region data between two-stage and three-stage analysis. Different colors represent each category, with numbers indicating intersections.  B) A bar graph shows counts of feature types across cytosine and region categories. Colors represent promoters, genes, downstream, transposable elements (TE), and intergenic regions, with counts labeled on each bar.  C) A bar graph shows transposable element (TE) counts across different genomic contexts: promoter, gene, and downstream. Multiple colors represent different TE classes, with a legend indicating each class.</alt-text>
</graphic>
</fig>
<p>In order to understand the potential functional roles of relevant features, we looked at their genomic context (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>. See <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S9</bold>
</xref> for the 3-stages model). From a total of 297 cytosines and 535 regions relevant for the 2-stages model, many of the features are found in transposable elements (TEs), with 127 cytosines and 255 regions located in TEs. Then, the following most frequent context is promoters, with 71 cytosines and 108 regions, followed by gene bodies, which harbor 51 cytosines and 96 regions.&#xa0;In the downstream context, we found 37 cytosines and 56 regions, whereas intergenic space contributes the fewest number of features, harboring 11 cytosines and 20 regions. In every context, region-level counts exceed cytosine-level counts.</p>
<p>Our pipeline identified a total of 52.79% transposable elements (TE) in the <italic>P. avium</italic> cv. Tieton genome. With that curated annotation, TEs with features were classified by genomic context and class (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5C</bold>
</xref>. See <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S10</bold>
</xref> for 3-stages model). From our TE annotation, 19 classes (including unknown) were found distributed among promoters, gene bodies, and downstream regions. Even though features from the 2-stages model were found in intergenic regions, none of the TEs with features are located in that genomic context (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S4</bold>
</xref>). Regarding TE classes, and besides unknown elements, two classes were enriched in each of the genomic contexts: LTR/ty3-retrotransposons, and LTR/Copia. Overall, the data indicates a consistent enrichment of LTR-type across regulatory (promoter), coding (gene), and proximal downstream segments, with fewer contributions from other TE classes. LTR/ty3-retrotransposons elements cluster toward pericentromeric or proximal regions of several chromosomes, most conspicuously on chromosomes 1, 4, and 5 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S11</bold>
</xref>). LTR/Copia TEs, by contrast, are more dispersed along chromosomal arms. Conversely, chromosomes 3, 6, 7, and 8 exhibit comparatively sparse LTR coverage.</p>
<p>Features from the model were later visualized to address their distribution along the eight chromosomes of <italic>Prunus avium</italic> (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>). Chromosome 1 exhibits the highest overall feature density. A thorough examination of the chromosome reveals an abundance of region counts, with some regions exhibiting values exceeding 20 megabase (Mb). This observation is further complemented by the presence of a significant number of cytosines. Chromosomes 3, 6, and 7 exhibit low overall region densities. The distribution of chromosome 4 features counts primarily located within the range of 25 to 32 Mb, and exhibiting windows that harbor both feature types. Chromosome 5 exhibits a distinctive mid-chromosomal high density at 10&#x2013;15 Mb, with a high number of cytosine counts, and several regions. Chromosome 8 exhibits a limited number of windows with regions and reduced cytosine densities. The colocalization of cytosines and regions in high-density within the chromosomes can be observed, including chromosome 1 at 0&#x2013;2 Mb and chromosome 5 at 10&#x2013;15 Mb. Overall, features are distributed along the genome with some cytosines clustering within genomic regions of interest such as previously reported QTLs in <italic>P. avium</italic>. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref> shows the colocalization between model features and QTLs associated with flowering date, temperature requirements (chill and heat), and maturity date. These QTLs were all identified in chromosome 4 of <italic>P. avium</italic> cv. Tieton, highlighting the relevance of linkage group 4 by carrying several QTLs associated with dormancy related traits.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Genomic distribution of features from the 2-stages model and associated QTLs. <bold>(A)</bold> The Manhattan plot displays the density of features across the eight chromosomes of P. avium, binned in 1 Mb intervals. The distribution of cytosine-level features is shown as orange dots, representing the counts of individual cytosines per 1 Mb window. Region-level features are depicted as blue triangles, indicating the number of regions per 1 Mb window. The corresponding chromosomes and genomic positions are indicated on the X axis as megabases (Mb). <bold>(B)</bold> Genomic distribution of features along P. avium chromosome 4. Each track position is based on the Tieton v2.0 genome annotation, in megabases. QTLs associated with flowering date (FL), chilling requirements (CR), and heat requirements (HR), in either RxG (Regina x Garnet) or RxL (Regina x Lapins) progenies were reported by <xref ref-type="bibr" rid="B11">Cast&#xe8;de et&#xa0;al. (2015)</xref>; and maturity date (MD) reported by <xref ref-type="bibr" rid="B10">Calle and W&#xfc;nsch (2020)</xref>. Blue lines indicating region features, and orange lines indicating single cytosine features. Cytosines and regions shorter than 20 kb were expanded to this width for improved visualization.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1659345-g006.tif">
<alt-text content-type="machine-generated">Chart A displays feature counts across chromosomes, with blue triangles for regions and orange circles for cytosines. Chart B shows QTLs, regions, and cytosines on Chromosome 4 by genomic position, with color-coded labels for QTL types.</alt-text>
</graphic>
</fig>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>The accurate and timely assessment of dormancy stages is a critical challenge in perennial horticulture, directly impacting orchard management and yield optimization. In this study, high-resolution DNA methylation profiling and machine learning techniques were integrated to identify robust epigenetic markers for predicting dormancy in sweet cherry. In plants, methylation occurs in three different contexts that serve a variety of roles in mediating environmental signals with molecular regulation. mCG methylation is found throughout the genome, including gene bodies, promoters, and repetitive regions, while mCHG and mCHH methylation are particularly enriched in repetitive sequences, including transposable elements, and play a key role in silencing these elements to maintain genome stability.</p>
<p>The results showed that in the three-class scenario, when methylated cytosines data were used, the paradormancy stage was misclassified 100% of the time by the RF model and 78% of the time by XGBoost. In contrast, endodormancy and ecodormancy had misclassification rates of 0% and 31% with RF, and 25% and 50% with XGBoost, respectively. Similar results were observed in methylated regions, where paradormancy exhibited lower classification efficiency compared to endodormancy and ecodormancy. This pattern may be explained by the imbalance in the number of samples per class, which appears to correlate with the error rate, with paradormancy having the fewest samples (9) and endodormancy having the most (36). Due to this class imbalance, both models tended to misclassify paradormancy samples, often assigning them to the majority class, endodormancy. The poor predictive performance in classifying unbalanced datasets can be since machine learning algorithms are designed to maximize overall accuracy and may struggle to accurately classify the minority class (<xref ref-type="bibr" rid="B23">Imani et&#xa0;al., 2025</xref>). The results of this study showed that RF outperformed XGBoost in accuracy and efficiency across all classes, regardless of class imbalance. Similar findings were reported by <xref ref-type="bibr" rid="B18">Dube and Verster (2023)</xref>, who observed that RF demonstrated robustness to class imbalance and outperformed several machine learning models, including na&#xef;ve Bayes, XGBoost, and k-nearest neighbors, which struggled with imbalanced data.</p>
<p>Feature selection applied to methylated cytosines and regions improved classification accuracy by approximately 40% for RF and 34% for XGBoost, highlighting the effectiveness of this approach. Similar findings were reported by <xref ref-type="bibr" rid="B22">Hassan et&#xa0;al. (2023)</xref>, where feature selection improved the accuracy of all classification models. This improvement is due to the effective selection of features, which ensures that the most relevant information is extracted from the dataset, enabling more accurate classification (<xref ref-type="bibr" rid="B25">Javidan et&#xa0;al., 2024</xref>). <xref ref-type="bibr" rid="B6">Bol&#xf3;n-Canedo and Alonso-Betanzos (2019)</xref> pointed out that different feature selection methods can yield varying subsets from the same dataset; thus, the results may lack stability. Therefore, combining multiple feature selection techniques, rather than relying on a single method, helps control variance, mitigates the limitations of individual approaches, and offers a more comprehensive view of feature relevance, which makes this approach successful (<xref ref-type="bibr" rid="B6">Bol&#xf3;n-Canedo and Alonso-Betanzos, 2019</xref>). Therefore, the approach presented in this study combines three algorithms for identifying relevant variables, since the combination provides a better approximation based on the idea that &#x201c;two heads are better than one&#x201d; (<xref ref-type="bibr" rid="B6">Bol&#xf3;n-Canedo and Alonso-Betanzos, 2019</xref>). Consistent with the classification results, RF outperformed XGBoost in accuracy following feature selection. Additionally, using methylated cytosine data proved more effective than using methylated regions, improving classification accuracy by 11% in RF and 19% in XGBoost. Similarly, <xref ref-type="bibr" rid="B22">Hassan et&#xa0;al. (2023)</xref> also showed that RF consistently outperformed XGBoost, both on the full dataset and after feature selection. Notably, the improvement in the classification after feature selection was also reflected in the t-SNE plots, since the clustering with the original data set, samples tended to cluster by cultivar rather than dormancy stage, whereas after feature selection, the clustering aligned more clearly with dormancy stages.</p>
<p>Despite the overall increase in accuracy after feature selection, the classification error for paradormancy samples remained high, ranging from 33% (with methylated cytosines using RF) to 78% (with methylated regions using XGBoost). This highlights a significant challenge in accurately classifying the paradormancy stage compared to endodormancy and ecodormancy. These results suggest that neither RF nor XGBoost was able to overcome the issue of class imbalance, even after applying feature selection. It is worth noting that the SMOTE algorithm, which is designed to generate synthetic examples of the minority class and alleviate class imbalance (<xref ref-type="bibr" rid="B20">Fern&#xe1;ndez et&#xa0;al., 2018</xref>), was also tested (data not shown). However, its application did not lead to improved classification performance. As a result, in the second scenario, paradormancy was excluded from the dataset, and only endodormancy and ecodormancy were used to train and build the classification models.</p>
<p>In the second scenario, where only endodormancy and ecodormancy were considered, the RF model showed an increase in accuracy of 15% with methylated cytosines and 9% with methylated regions, compared to the three-class scenario. In contrast, XGBoost experienced a decline in accuracy of 24% and 18% for cytosines and regions, respectively. Consistent with the second scenario observations, both models showed improved accuracy when feature selection was applied, reinforcing the importance of identifying the most informative variables. However, it is important to note that the observed increase in accuracy does not necessarily reflect an improvement in the overall classification performance, since in the three-class scenario, much of the accuracy lost was tied to the misclassification of paradormancy samples. In both scenarios, RF consistently achieved the highest accuracy across all configurations, confirming its superior performance and greater stability under both balanced and imbalanced conditions. These findings further emphasize the limitations of XGBoost when dealing with skewed datasets, as well as the benefit of applying feature selection to enhance the model performance of dormancy stages classification based on epigenetic information.</p>
<p>Given that RF consistently outperformed XGBoost in accuracy and robustness across all scenarios, interpretability efforts based on SHAP focused on RF models trained with the informative feature&#xa0;sets of both cytosine-level and region-level methylation data. In the cytosine dataset, features such as chr_4_31092165 and chr_1_38152240 emerged as the most impactful, particularly for distinguishing the ecodormancy stage. This suggests that methylation at these loci could be tightly associated with epigenetic regulation mechanisms unique to ecodormancy. Other features, like chr_3_23416192 and chr_4_31050901, displayed stronger relevance for paradormancy, indicating that although paradormancy was difficult to classify due to data imbalance, certain epigenetic markers do exist that could potentially improve its prediction if more balanced data were available. Similarly, in the regional methylation dataset, regions like chr_3_9331371_9331488 and chr_1_27137350_27137417 had high SHAP values. Interestingly, some regions, such as chr_6_9727681_9727971, demonstrated relatively balanced contributions across all three dormancy stages. This may indicate shared or transitional roles in the broader regulation of dormancy, potentially pointing to core epigenetic signatures common to different dormancy stages. The individual SHAP plots for each dormancy stage (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S8</bold>
</xref>) showed that features such as chr_1_38152240 contributed strongly and positively to the correct classification of ecodormancy samples, while exhibiting negative contributions in misclassified samples labeled as endodormancy. This highlights the high discriminative power of this feature in separating these two stages. In contrast, features like chr_3_8464601, although showing high SHAP values, presented similar contributions in both ecodormancy and endodormancy samples. This suggests that it may not be informative on its own but could contribute to the model&#x2019;s predictions through interactions with other features, an effect often captured by ensemble models like SHAP (<xref ref-type="bibr" rid="B47">&#x160;trumbelj and Kononenko, 2014</xref>).</p>
<p>There are several examples in fruit trees where methylation of individual cytosines plays key roles during dormancy. <xref ref-type="bibr" rid="B30">Kumar et&#xa0;al. (2016)</xref> examined four developmental stages in apple, including dormant buds, and showed, using bisulfite&#x2013;sequenced MSAP fragments, that the number of methylated cytosines at single loci correlated with transcript levels of those genes (e.g., an acid phosphatase 1-like gene and a galactose oxidase-like gene). <xref ref-type="bibr" rid="B41">Prudencio et&#xa0;al. (2018)</xref> in almond found that over 90 % of cytosines are unmethylated, while approximately 1-1.3 % are fully methylated in CpG contexts; these polymorphic 5-mC sites were conserved across two years and between dormant and non&#x2013;dormant buds. These studies demonstrate that methylation at even a handful of cytosines in specific genomic contexts can meaningfully modulate gene expression. Given their stability and stage specificity, single&#x2013;cytosine marks hold considerable promise as robust biomarkers for dormancy transitions. Although the precise mechanism by which a single methylated cytosine influences gene expression remains unclear, we hypothesize that such effects occur when the cytosine resides within functionally relevant regions, particularly transcription factor binding sites or cis&#x2013;regulatory modules.</p>
<p>Differences in the number of relevant features were evident between the dormancy models. The 2-stages model generated the largest and most unique feature sets, resulting in 500 regions and 283 cytosine unique features. This genome-wide representation is consistent with the higher predictive power of 2-stage cytosine methylation data in Random Forest and XGBoost, suggesting that the additional loci due to the higher input samples, supplies informative data that improves the classification. By contrast, the 3-stage model contributed a leaner, yet distinct, repertoire of 182 unique regions and 58 unique cytosines, which may represent the early dormancy stage paradormancy signatures that the 2-stage model misses.</p>
<p>The reduced number of individual cytosine overlapping regions in the 2-stages model (4) may imply that single cytosines show more sensitivity to coverage noise or local sequence context, whereas regions provide a more robust, though less precise, relevant dormancy-associated loci. In order to address those questions, we searched for the genomic context of those relevant features from the 2-stage model to gather insights about the functional roles of those cytosines (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>). The genomic context frequencies rank in the same order across both feature types. From highest to lowest: TEs, promoter, gene bodies, downstream, and intergenic, indicating that cytosine and region features share similar genomic preferences. Since TEs were the highest genomic context with features, and they can be located in any of the above genomic contexts, we further study the genomic context of TEs with features and their corresponding class (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>). The number of TEs that remained unclassified is still high for <italic>P. avium</italic> Tieton annotation, highlighting the need for cultivar-specific transposable elements annotation.</p>
<p>Certain TE families exhibit preference for specific genomic features. A well known example is ONSEN, an LTR/Copia element in <italic>Arabidopsis thaliana</italic> that integrates preferentially within genes (<xref ref-type="bibr" rid="B36">Merkulov et&#xa0;al., 2025</xref>). Members of the LTR/Copia class often insert near gene-rich regions, unlike other TEs that target intergenic areas. Several Copia families including Copia87/ONSEN (<xref ref-type="bibr" rid="B24">Ito et&#xa0;al., 2011</xref>), COPIA37, TERESTRA, and ROMANIAT5 (<xref ref-type="bibr" rid="B40">Pietzenuk et&#xa0;al., 2016</xref>) have been well described as stress-responsive TEs, becoming active when temperature rises due to heat-responsive elements in their promoter regions. The LTR/Copia and Ty3, flanked by a promoter region, can undergo a replication cycle (<xref ref-type="bibr" rid="B38">Oberlin et&#xa0;al., 2017</xref>), with the later ability to confer neighboring genes the responsiveness to abiotic stressors (<xref ref-type="bibr" rid="B49">Xu et&#xa0;al., 2024</xref>). LTR/ty3-retrotransposons elements in <italic>P. avium</italic> were found predominantly in heterochromatic regions and LTR/Copia elements located in euchromatic domains, consistent with observations in almond (<xref ref-type="bibr" rid="B1">Alioto et&#xa0;al., 2020</xref>). TE activity is often suppressed through epigenetic mechanisms such as the RNA-directed DNA methylation (RdDM) pathway, which is responsive to environmental conditions. <xref ref-type="bibr" rid="B37">Nozawa et&#xa0;al. (2022)</xref> reported that ONSEN expression varies across <italic>Arabidopsis</italic> ecotypes depending on DNA methylation levels, with the Kyoto ecotype showing increased expression due to reduced methylation conferring release from transcriptional silencing. Although the functional dynamics of TE regulation during dormancy remain unclear in sweet cherry, we hypothesize that small RNAs involved in RdDM, previously characterized in <italic>P. avium</italic> (<xref ref-type="bibr" rid="B43">Rothkegel et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B44">2020</xref>; <xref ref-type="bibr" rid="B29">Kuhn et&#xa0;al., 2025</xref>), may regulate TE transcription and potentially modulate the expression of adjacent genes involved in dormancy transitions.</p>
<p>Mapping these loci along the eight chromosomes of <italic>P. avium</italic> (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>) revealed broad genome coverage, with particularly relevant regions on chromosome 1 (0&#x2013;2 Mb), chromosome 4 (25&#x2013;32 Mb), and chromosome 5 (10&#x2013;15 Mb). Chromosome 4 (LG4, <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>) includes a well reported QTL hotspot where flowering date (FD), chilling requirement (CR), and heat requirement (HR) QTLs, identified in sweet cherry progenies with Regina as the progenitor (Regina &#xd7; Lapins and Regina &#xd7; Garnet), co-localize with maturity date (MD) QTLs (<xref ref-type="bibr" rid="B12">Cast&#xe8;de et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B10">Calle and W&#xfc;nsch, 2020</xref>; <xref ref-type="bibr" rid="B9">Calle et&#xa0;al, 2020</xref>). Our model identified a concentration of informative regions, with fewer cytosine features, suggesting a broader chromatin-level regulation rather than single-site CpG variation may underlie dormancy transitions. This is consistent with previous reports demonstrating that DNA methylation and chromatin remodeling regulate dormancy release in <italic>Prunus</italic> (<xref ref-type="bibr" rid="B29">Kuhn et&#xa0;al., 2025</xref>; <xref ref-type="bibr" rid="B44">Rothkegel et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B55">Zhu et&#xa0;al., 2020</xref>). Within chromosome 4, several biologically relevant candidate genes have been reported (<xref ref-type="bibr" rid="B12">Cast&#xe8;de et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B10">Calle and W&#xfc;nsch, 2020</xref>). These include <italic>EMF2</italic>, a MADS-box transcription factor involved in floral meristem identity (<xref ref-type="bibr" rid="B51">Yoshida et&#xa0;al., 2001</xref>); <italic>NUA</italic> (Nuclear Pore Anchor), which regulates nuclear-cytoplasmic transport and chromatin organization; NAC domain transcription factors associated with stress responses and developmental control; <italic>EXPA1</italic> (Expansin A1), involved in cell wall loosening and bud outgrowth; bHLH (basic helix-loop-helix) transcription factors implicated in hormone and light signaling pathways; and WRKY transcription factors known for their role in stress-responsive gene regulation.</p>
<p>Additionally, a QTL on LG1 includes <italic>AGL24</italic>-like MADS-box genes, which are involved in floral transition and respond to environmental cues, like the <italic>DORMANCY-ASSOCIATED MADS-box</italic> (<italic>DAM</italic>) gene cluster (<italic>DAM1</italic> to <italic>DAM6</italic>) (<xref ref-type="bibr" rid="B11">Cast&#xe8;de et&#xa0;al., 2015</xref>), which also overlaps with regions selected by our model. These genes have been shown to play a central role in the repression of bud growth during dormancy and its release in response to chilling accumulation. <xref ref-type="bibr" rid="B12">Cast&#xe8;de et&#xa0;al. (2014)</xref> identified this region as a key determinant of phenological variation in sweet cherry, particularly in high-chill cultivars such as Regina. Additionally, <xref ref-type="bibr" rid="B10">Calle and W&#xfc;nsch (2020)</xref> demonstrated that this QTL on LG1 remains significant even in low-chill cultivars, supporting its broad importance across diverse genetic backgrounds. In our study, several methylation features selected by the model colocalized with this QTL region, suggesting that epigenetic regulation contributes to differential <italic>DAM</italic> genes expression among cultivars. Increasing evidence supports the role of DNA methylation in modulating <italic>DAM</italic> genes activity during dormancy transitions. <xref ref-type="bibr" rid="B55">Zhu et&#xa0;al. (2020)</xref> showed that changes in methylation levels at the promoters and gene bodies of <italic>DAM</italic> genes in peach were associated with their transcriptional downregulation during chilling accumulation. Specifically, DNA hypomethylation correlated with a reduction in <italic>DAM</italic> transcript levels, suggesting that chilling-induced epigenetic remodeling facilitates dormancy release. Similar patterns of dynamic methylation have been reported in other <italic>Prunus</italic> species, underscoring the conserved nature of this regulatory mechanism. Our results align with these findings and point to DNA methylation as a key regulatory layer influencing the dormancy-related function of <italic>DAM</italic> genes. These observations support a model in which genetic loci like <italic>DAM</italic> are subject to cultivar-specific epigenetic modulation, contributing to phenotypic diversity in dormancy behavior across sweet cherry germplasm. Altogether, these results show the relevance of the LG1 QTL and the <italic>DAM</italic> cluster as conserved regulators of dormancy, while also highlighting the potential role of epigenetic variability in shaping cultivar-specific responses to chilling.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>This study presents a new integrating frame that combines the high-resolution DNA methylation profile with automatic learning&#xa0;approaches to classify dormancy stages in <italic>Prunus avium</italic>. The results showed that the methylation data at the cytosine level, when processed through the selection of features and interpreted through form values, provide highly informative epigenetic markers to distinguish endodormancy and ecodormancy stages with high precision. Among the proven models, RF constantly exceeded XGBoost in robustness, precision, and interpretability in balanced and unbalanced scenarios. It is important to note that some of the methylation features co-localize with related QTLs previously identified by other groups, such as chilling and heat requirement flowering date and maturity date. The results of this study contribute a valuable methodological and conceptual advance for epigenetic research in these traits. It provides a fundamental resource for future functional validation studies and prepares the scenario to develop predictive tools in horticultural management. When discovering key methylation markers and their genomic contexts, this study improves our understanding of dormancy regulation and underlines the importance of epigenetic mechanisms in the adaptation and resistance of perennial crops.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>GS: Formal analysis, Investigation, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. PP: Data curation, Formal analysis, Software, Visualization, Writing &#x2013; original draft. CU: Formal analysis, Validation, Writing &#x2013; original draft. JG-L: Formal analysis, Validation, Writing &#x2013; original draft. CM: Funding acquisition, Methodology, Software, Supervision, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. AMA: Conceptualization, Funding acquisition, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This research was funded by Agencia Nacional de Investigaci&#xf3;n y Desarrollo (ANID/Chile) FONDECYT 1230163 (AMA), ANID/FONDECYT 11240273 (CM), ANID/ACT210007 (AMA and CM) and Universidad Mayor Doctoral scholarship (GS).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The authors thank Franco Aguirre and Dr. Giovana Acha for the DNA extraction of samples from experiments 2 and 3. We also thank Agr&#xed;cola Garc&#xe9;s to give us access to their sweet cherry productive fields.</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors&#xa0;and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2025.1659345/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2025.1659345/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alioto</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Alexiou</surname> <given-names>K. G.</given-names>
</name>
<name>
<surname>Bardil</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Barteri</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Castanera</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Cruz</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Transposons played a major role in the diversification between the closely related almond and peach genomes: results from the almond genome sequence</article-title>. <source>Plant J.</source> <volume>101</volume>, <fpage>455</fpage>&#x2013;<lpage>472</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/tpj.14538</pub-id>, PMID: <pub-id pub-id-type="pmid">31529539</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alonso-Gonz&#xe1;lez</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Gutmann</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Aalstad</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Fayad</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Gascoin</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Snowpack dynamics in the Lebanese mountains from quasi-dynamically downscaled ERA5 reanalysis updated by assimilating remotely-sensed fractional snow-covered area</article-title>. <source>Hydrol. Earth Syst. Sci.</source> <volume>25</volume>, <fpage>1</fpage>&#x2013;<lpage>31</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5194/hess-25-4455-2021</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Andrews</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). <source>FastQC: A Quality Control Tool for High Throughput Sequence Data</source>. Available online at: <uri xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/fastqc/">http://www.bioinformatics.babraham.ac.uk/projects/fastqc/</uri> (Accessed <access-date>September 20, 2024</access-date>).</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baumgarten</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zohner</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>Gessler</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Vitasse</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Chilled to be forced: The best dose to wake up buds from winter dormancy</article-title>. <source>New Phytol.</source> <volume>230</volume>, <fpage>1366</fpage>&#x2013;<lpage>1377</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/nph.17270</pub-id>, PMID: <pub-id pub-id-type="pmid">33577087</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bi</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xiang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An interpretable prediction model for identifying N7-methylguanosine sites based on XGboost and SHAP</article-title>. <source>Mol. Ther.</source> <volume>22</volume>, <fpage>362</fpage>&#x2013;<lpage>372</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.omtn.2020.08.022</pub-id>, PMID: <pub-id pub-id-type="pmid">33230441</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bol&#xf3;n-Canedo</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Alonso-Betanzos</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Ensembles for feature selection: A review and future trends</article-title>. <source>Inf. Fusion</source> <volume>52</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inffus.2018.11.008</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Bagging predictors</article-title>. <source>Mach. Learn.</source> <volume>24</volume>, <fpage>123</fpage>&#x2013;<lpage>140</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/BF00058655</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bzdok</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Altman</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Martin</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Statistics versus machine learning</article-title>. <source>Nat. Methods</source> <volume>15</volume>, <fpage>233</fpage>&#x2013;<lpage>234</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nmeth.4642</pub-id>, PMID: <pub-id pub-id-type="pmid">30100822</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calle</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Iezzoni</surname> <given-names>A.</given-names>
</name>
<name>
<surname>W&#xfc;nsch</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Genetic dissection of bloom time in low chilling sweet cherry (<italic>Prunus avium</italic> L.) using a multi-family QTL approach</article-title>. <source>Front. Plant Sci.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2019.01647</pub-id>, PMID: <pub-id pub-id-type="pmid">31998337</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calle</surname> <given-names>A.</given-names>
</name>
<name>
<surname>W&#xfc;nsch</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Multiple-jpopulation QTL mapping of maturity and fruit-quality traits reveals LG4 region as a breeding target in sweet cherry (<italic>Prunus avium</italic> L.)</article-title>. <source>Hortic. Res.</source> <volume>7</volume>, <fpage>127</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41438-020-00349-2</pub-id>, PMID: <pub-id pub-id-type="pmid">32821410</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cast&#xe8;de</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Campoy</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Le Dantec</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Quero-Garcia</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Barreneche</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wenden</surname> <given-names>B.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>Mapping of candidate genes involved in bud dormancy and flowering time in sweet cherry (<italic>Prunus avium</italic>)</article-title>. <source>PloS One</source> <volume>10</volume>, <fpage>e0143250</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0143250</pub-id>, PMID: <pub-id pub-id-type="pmid">26587668</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cast&#xe8;de</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Campoy</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Quero-Garc&#xed;a</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Le Dantec</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Lafargue</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Barreneche</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>Genetic determinism of phenological traits highly affected by climate change in <italic>Prunus avium</italic>: flowering date dissected into chilling and heat requirements</article-title>. <source>New Phytol.</source> <volume>202</volume>, <fpage>703</fpage>&#x2013;<lpage>715</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/nph.12658</pub-id>, PMID: <pub-id pub-id-type="pmid">24417538</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>XGBoost: A scalable tree boosting system</article-title>,&#x201d; in KDD '<source>16: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. (<publisher-loc>New York, NY, United States</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>785</fpage>&#x2013;<lpage>794</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id>
</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chu</surname> <given-names>Y. R.</given-names>
</name>
<name>
<surname>Jo</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>G. E.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>C. H.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>D. J.</given-names>
</name>
<name>
<surname>Che</surname> <given-names>S. H.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Non-destructive seed viability assessment via multispectral imaging and stacking ensemble learning</article-title>. <source>Agriculture</source> <volume>14</volume>, <fpage>1679</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture14101679</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cline</surname> <given-names>M. G.</given-names>
</name>
<name>
<surname>Deppong</surname> <given-names>D. O.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>The role of apical dominance in paradormancy of temperate woody plants: a reappraisal</article-title>. <source>J. Plant Physiol.</source> <volume>155</volume>, <fpage>350</fpage>&#x2013;<lpage>356</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0176-1617(99)80116-3</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Considine</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Considine</surname> <given-names>J. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>On the language and physiology of dormancy and quiescence in plants</article-title>. <source>J. Exp. Bot.</source> <volume>67</volume>, <fpage>3189</fpage>&#x2013;<lpage>3203</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jxb/erw138</pub-id>, PMID: <pub-id pub-id-type="pmid">27053719</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Pandey</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Perales</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Allona</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M. R. I.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Molecular advances in bud dormancy in trees</article-title>. <source>J. Exp. Bot.</source> <volume>75</volume>, <fpage>6063</fpage>&#x2013;<lpage>6075</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jxb/erae183</pub-id>, PMID: <pub-id pub-id-type="pmid">38650362</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dube</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Verster</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Enhancing classification performance in imbalanced datasets: A comparative analysis of machine learning models</article-title>. <source>Data Sci. Finance Econom.</source> <volume>3</volume>, <fpage>354</fpage>&#x2013;<lpage>379</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3934/DSFE.2023021</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fad&#xf3;n</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Herrero</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rodrigo</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Flower development in sweet cherry framed in the BBCH scale</article-title>. <source>Sci. Hortic.</source> <volume>192</volume>, <fpage>141</fpage>&#x2013;<lpage>147</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scienta.2015.05.027</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fern&#xe1;ndez</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Garc&#xed;a</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Herrera</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Chawla</surname> <given-names>N. V.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>SMOTE for Learning From Imbalanced Data: Progress and Challenges, Marking the 15-Year Anniversary</article-title>. <source>J. Artif. Intell. Res.</source> <volume>61</volume>, <fpage>863</fpage>&#x2013;<lpage>905</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1613/jair.1.11192</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flynn</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Hubley</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Goubert</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Rosen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Clark</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Feschotte</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>RepeatModeler2 for automated genomic discovery of transposable element families</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>117</volume>, <fpage>9451</fpage>&#x2013;<lpage>9457</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1921046117</pub-id>, PMID: <pub-id pub-id-type="pmid">32300014</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hassan</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Hassan</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Yasmin</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M. A. R.</given-names>
</name>
<name>
<surname>Zaman</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>K. K.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>A comparative assessment of machine learning algorithms with the Least Absolute Shrinkage and Selection Operator for breast cancer detection and prediction</article-title>. <source>Decision Analytics J.</source> <volume>7</volume>, <fpage>100245</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.dajour.2023.100245</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Imani</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Beikmohammadi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Arabnia</surname> <given-names>H. R.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Comprehensive analysis of random Forest and XGBoost performance with SMOTE, ADASYN, and GNUS under varying imbalance levels</article-title>. <source>Technologies</source> <volume>13</volume>, <fpage>88</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/technologies13030088</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ito</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Gaubert</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Bucher</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Mirouze</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vaillant</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Paszkowski</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>An siRNA pathway prevents transgenerational retrotransposition in plants subjected to stress</article-title>. <source>Nature</source> <volume>472</volume>, <fpage>115</fpage>&#x2013;<lpage>119</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature09861</pub-id>, PMID: <pub-id pub-id-type="pmid">21399627</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Javidan</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Banakar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Rahnama</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Vakilian</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Ampatzidis</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Feature engineering to identify plant diseases using image processing and artificial intelligence: A comprehensive review</article-title>. <source>Smart Agric. Technol.</source> <volume>8</volume>, <elocation-id>100480</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.ATECH.2024.100480</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kavzoglu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Teke</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predictive performances of ensemble machine learning algorithms in landslide susceptibility mapping using random forest, extreme gradient boosting (XGBoost) and natural gradient boosting (NGBoost)</article-title>. <source>Arab. J. Sci. Eng.</source> <volume>47</volume>, <fpage>7367</fpage>&#x2013;<lpage>7385</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s13369-022-06560-8</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krueger</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Andrews</surname> <given-names>S. R.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Bismark: a flexible aligner and methylation caller for Bisulfite-Seq applications</article-title>. <source>Bioinformatics</source> <volume>27</volume>, <fpage>1571</fpage>&#x2013;<lpage>1572</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btr167</pub-id>, PMID: <pub-id pub-id-type="pmid">21493656</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Krueger</surname> <given-names>F.</given-names>
</name>
<name>
<surname>James</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Ewels</surname> <given-names>P. E.</given-names>
</name>
<name>
<surname>E</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Weinstein</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Schuster-Boeckler</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Felixkrueger/Trimgalore: V0.6.10</source>. Available online at: <uri xlink:href="https://github.com/Felixkrueger/Trimgalore">https://github.com/Felixkrueger/Trimgalore</uri> (Accessed <access-date>September 20, 2024</access-date>).</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuhn</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Arellano</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ponce</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Hodar</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Correa</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Multari</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>RNA-seq and WGBS analyses during fruit ripening and in response to ABA in sweet cherry (<italic>Prunus avium</italic>) reveal genetic and epigenetic modulation of auxin and cytokinin genes</article-title>. <source>J. Plant Growth Regul.</source> <volume>44</volume>, <fpage>1165</fpage>&#x2013;<lpage>1187</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00344-024-11340-9</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Rattan</surname> <given-names>U. K.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>A. K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Chilling-mediated DNA methylation changes during dormancy and its release reveal the importance of epigenetic regulation during winter dormancy in apple (<italic>Malus</italic> x <italic>domestica</italic> borkh.)</article-title>. <source>PloS One</source> <volume>11</volume>, <fpage>e0149934</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0149934</pub-id>, PMID: <pub-id pub-id-type="pmid">26901339</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Changes in physiological and biochemical properties and variation in DNA methylation patterns during dormancy and dormancy release in blueberry (<italic>Vaccinium corymbosum</italic> L.)</article-title>. <source>Plant Physiol. J.</source> <volume>51</volume>, <fpage>1133</fpage>&#x2013;<lpage>1141</lpage>.</citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Research on factors affecting global grain legume yield based on explainable artificial intelligence</article-title>. <source>Agriculture</source> <volume>14</volume>, <fpage>438</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture14030438</pub-id>
</citation></ref>
<ref id="B33">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>New machine learning algorithm: Random forest</article-title>,&#x201d; in <conf-name>Information Computing and Applications: Third International Conference, ICICA 2012</conf-name>, <conf-loc>Chengde, China</conf-loc>, <conf-date>September 14-16, 2012</conf-date>. <fpage>246</fpage>&#x2013;<lpage>252</lpage> (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luedeling</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Girvetz</surname> <given-names>E. H.</given-names>
</name>
<name>
<surname>Semenov</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>P. H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Climate change affects winter chill for temperate fruit and nut trees</article-title>. <source>PloS One</source> <volume>6</volume>, <fpage>e20155</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0020155</pub-id>, PMID: <pub-id pub-id-type="pmid">21629649</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>S.-I.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>A unified approach to interpreting model predictions</article-title>,&#x201d; in <source>Proceedings of the 31st International Conference on Neural Information Processing Systems</source> (<publisher-name>Curran Associates</publisher-name>, <publisher-loc>Long Beach, CA</publisher-loc>), <fpage>4766</fpage>&#x2013;<lpage>4775</lpage>.</citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Merkulov</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Latypova</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Tiurin</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Serganova</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kirov</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>DNA methylation and alternative splicing safeguard genome and transcriptome after a retrotransposition burst in arabidopsis thaliana</article-title>. <source>Int. J. Mol. Sci.</source> <volume>26</volume>, <fpage>4816</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ijms26104816</pub-id>, PMID: <pub-id pub-id-type="pmid">40429956</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nozawa</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Masuda</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Saze</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Ikeda</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Suzuki</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Takagi</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Epigenetic regulation of ecotype-specific expression of the heat-activated transposon ONSEN</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>899105</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.899105</pub-id>, PMID: <pub-id pub-id-type="pmid">35923888</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oberlin</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sarazin</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chevalier</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Voinnet</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Mari-Ordonez</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A genome-wide transcriptome and translatome analysis of arabidopsis transposons identifies a unique and conserved genome expression strategy for Ty1/Copia retroelements</article-title>. <source>Genome Res.</source> <volume>27</volume>, <fpage>1549</fpage>&#x2013;<lpage>1562</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/gr.220723.117</pub-id>, PMID: <pub-id pub-id-type="pmid">28784835</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>G. I.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Model multifactor analysis of soil heavy metal pollution on plant germination in Southeast Chengdu, China: Based on redundancy analysis, factor detector, and XGBoost-SHAP</article-title>. <source>Sci. Total. Environ.</source> <volume>954</volume>, <fpage>176605</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scitotenv.2024.176605</pub-id>, PMID: <pub-id pub-id-type="pmid">39349201</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pietzenuk</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Markus</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gaubert</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Bagwan</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Merotto</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Bucher</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Recurrent evolution of heat-responsiveness in Brassicaceae COPIA elements</article-title>. <source>Genome Biol.</source> <volume>17</volume>, <fpage>209</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-016-1072-3</pub-id>, PMID: <pub-id pub-id-type="pmid">27729060</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prudencio</surname> <given-names>&#xc1;.S.</given-names>
</name>
<name>
<surname>Werner</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Mart&#xed;nez-Garc&#xed;a</surname> <given-names>P. J.</given-names>
</name>
<name>
<surname>Dicenta</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Ros</surname> <given-names>R. M.</given-names>
</name>
<name>
<surname>Mart&#xed;nez-G&#xf3;mez</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DNA methylation analysis of dormancy release in almond (<italic>Prunus dulcis</italic>) flower buds using epi-genotyping by sequencing</article-title>. <source>Int. J. Mol. Sci.</source> <volume>19</volume>, <elocation-id>3542</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ijms19113542</pub-id>, PMID: <pub-id pub-id-type="pmid">30423798</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raihan</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M. A.-M.</given-names>
</name>
<name>
<surname>Kee</surname> <given-names>S.-H.</given-names>
</name>
<name>
<surname>Nahid</surname> <given-names>A.-A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Detection of&#xa0;the&#xa0;chronic kidney disease using xgboost classifier and explaining the influence of the&#xa0;attributes on the model using shap</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>6263</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-023-33525-0</pub-id>, PMID: <pub-id pub-id-type="pmid">37069256</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rothkegel</surname> <given-names>K.</given-names>
</name>
<name>
<surname>S&#xe1;nchez</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Montes</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Greve</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Tapia</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bravo</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>DNA methylation and small interference RNAs participate in the regulation of MADS-box genes involved in dormancy in sweet cherry (<italic>Prunus avium</italic> L.)</article-title>. <source>Tree Physiol.</source> <volume>37</volume>, <fpage>1739</fpage>&#x2013;<lpage>1751</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/treephys/tpx055</pub-id>, PMID: <pub-id pub-id-type="pmid">28541567</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rothkegel</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Sandoval</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Soto</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Lissette</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Riveros</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Lillo</surname> <given-names>V.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Dormant but active: chilling accumulation modulates the epigenome and transcriptome of Prunus avium during bud dormancy</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2020.01115</pub-id>, PMID: <pub-id pub-id-type="pmid">32765576</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Smit</surname> <given-names>A. F. A.</given-names>
</name>
<name>
<surname>Hubley</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Green</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <source>2013&#x2013;2015. RRepeatMasker Open-4.0</source>. Available online at: <uri xlink:href="http://www.repeatmasker.org">http://www.repeatmasker.org</uri>.</citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soto</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Sanchez</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Nu&#xf1;ez</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Montes</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Rothkegel</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Andrade</surname> <given-names>P.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>).&#xa0;<article-title>Small RNA differential expression analysis reveals miRNAs involved in dormancy progression in sweet cherry floral buds</article-title>. <source>Plants</source> <volume>11</volume>, <elocation-id>2396</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/plants11182396</pub-id>, PMID: <pub-id pub-id-type="pmid">36145795</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#x160;trumbelj</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Kononenko</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Explaining prediction models and individual predictions with feature contributions</article-title>. <source>Knowl. Inf. Syst.</source> <volume>41</volume>, <fpage>647</fpage>&#x2013;<lpage>665</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10115-013-0679-x</pub-id>
</citation></ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weinberger</surname> <given-names>J. H.</given-names>
</name>
</person-group> (<year>1950</year>). <article-title>Chilling requirements of peach varieties</article-title>. <source>Am. Soc. Hortic. Sci.</source> <volume>56</volume>, <fpage>122</fpage>&#x2013;<lpage>128</lpage>.</citation></ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Thieme</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Roulin</surname> <given-names>A. C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Natural diversity of heat-induced transcription of retrotransposons in Arabidopsis thaliana</article-title>. <source>Genome Biol. Evol.</source> <volume>16</volume>, <fpage>evae242</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/gbe/evae242</pub-id>, PMID: <pub-id pub-id-type="pmid">39523776</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Moriguchi</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Teng</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Bud endodormancy in deciduous fruit trees: advances and prospects</article-title>. <source>Hortic. Res.</source> <volume>8</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41438-021-00575-2</pub-id>, PMID: <pub-id pub-id-type="pmid">34078882</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoshida</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Yanai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Kato</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hiratsuka</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Miwa</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2001</year>). <article-title>EMBRYONIC FLOWER2, a novel polycomb group protein homolog, mediates shoot development and flowering in Arabidopsis</article-title>. <source>Plant Cell</source> <volume>13</volume>, <fpage>2471</fpage>&#x2013;<lpage>2481</lpage>., PMID: <pub-id pub-id-type="pmid">11701882</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Siddique</surname> <given-names>K. H.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Game analysis of future rice yield changes in China based on explainable machine-learning and planting date optimization</article-title>. <source>Field Crop Res.</source> <volume>317</volume>, <fpage>109557</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s13369-022-06560-8</pub-id>
</citation></ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhuo</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Transcriptome profiles reveal the crucial roles of hormone and sugar in the bud dormancy of Prunus mume</article-title>. <source>Sci. Rep.</source> <volume>8</volume>, <fpage>5090</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-018-23108-9</pub-id>, PMID: <pub-id pub-id-type="pmid">29572446</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Ng</surname> <given-names>H. K.</given-names>
</name>
<name>
<surname>Drautz-Moses</surname> <given-names>D. I.</given-names>
</name>
<name>
<surname>Schuster</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Beck</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Systematic evaluation of library preparation methods and sequencing platforms for high-throughput whole genome bisulfite sequencing</article-title>. <source>Sci. Rep.</source> <volume>9</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-019-46875-5</pub-id>, PMID: <pub-id pub-id-type="pmid">31316107</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>P. Y.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Dardick</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Callahan</surname> <given-names>A.</given-names>
</name>
<name>
<surname>An</surname> <given-names>Y. Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Thermal-responsive genetic and epigenetic regulation of DAM cluster controlling dormancy and chilling requirement in peach floral buds</article-title>. <source>Hortic. Res.</source> <volume>7</volume>, <fpage>114</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41438-020-0336-y</pub-id>, PMID: <pub-id pub-id-type="pmid">32821397</pub-id></citation></ref>
</ref-list>
</back>
</article>