<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. For. Glob. Change</journal-id>
<journal-title>Frontiers in Forests and Global Change</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. For. Glob. Change</abbrev-journal-title>
<issn pub-type="epub">2624-893X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/ffgc.2025.1599510</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Forests and Global Change</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Leveraging Sentinel-1/2 time series and deep learning for accurate forest tree species mapping</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Tan</surname> <given-names>Jun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3013999/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Li</surname> <given-names>Jing</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ma</surname> <given-names>Tianyue</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yan</surname> <given-names>Xingguang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Huo</surname> <given-names>Ziye</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Geoscience and Surveying Engineering, China University of Mining and Technology-Beijing</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>State Key Laboratory of Information Engineering in Surveying, Mapping and Remote Sensing, Wuhan University</institution>, <addr-line>Wuhan</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Fernando J. Aguilar, University of Almer&#x00ED;a, Spain</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Emilio Ram&#x00ED;rez-Juid&#x00ED;as, University of Seville, Spain</p><p>Rajesh Vanguri, Sapienza University of Rome, Italy</p></fn>
<corresp id="c001">&#x002A;Correspondence: Jing Li, <email>lijing@cumtb.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1599510</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>03</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Tan, Li, Ma, Yan and Huo.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Tan, Li, Ma, Yan and Huo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Accurate mapping of tree species is critical for forest management, carbon sequestration estimates and ecosystem assessment. Remote sensing provides an efficient approach using satellite image time series (SITS), but complex data poses challenges for classifiers and feature analysis. This study presents a deep learning-based classification method using Sentinel-1/2 SITS for mapping forest tree species and tree species biodiversity. Specifically, temporal data from unlabeled forest pixels were used for pretraining the model through self-supervised learning, followed by fine-tuning with species samples, enhancing model performance. Various configurations of temporal data were tested for classification, and their impact was evaluated. To address species maps accuracy overestimation caused by homogeneous pure-species stands, a pseudo-labeling approach was employed to incorporate mixed-species scenarios. Additionally, statistical and visualization methods were applied to SITS and model analysis. The results showed that longer time series tended to improve species identification and model confidence, with OA increasing from 0.496 (6&#x2013;7 months) to 0.795 (1&#x2013;12 months), macro-F1 from 0.384 to 0.779, and a significant improvement in predicted scores. As data from subsequent year was incorporated, accuracy growth slowed and stabilized, reaching OA of 0.847 and macro-F1 of 0.836, compared to 0.764 and 0.737 for the non-pretrained model. Certain vegetation indices, such as NDre and NDVIre, which are sensitive to physiological changes, highlight species differences during key phenological stages, especially between deciduous and evergreen species. This study demonstrates the potential of combining SITS with deep learning for species classification and provides a comprehensive analysis, contributing to ecological research and sustainable forest management.</p>
</abstract>
<kwd-group>
<kwd>tree species classification</kwd>
<kwd>remote sensing</kwd>
<kwd>deep learning</kwd>
<kwd>biodiversity</kwd>
<kwd>Sentinel-1/2</kwd>
</kwd-group>
<counts>
<fig-count count="13"/>
<table-count count="2"/>
<equation-count count="17"/>
<ref-count count="60"/>
<page-count count="19"/>
<word-count count="11726"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Temperate and Boreal Forests</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>1 Introduction</title>
<p>Forests are a key component of ecosystems, playing a crucial role in environmental sustainability and human wellbeing. Accurate mapping of forest tree species distribution and biodiversity is essential for forest management and conservation, ecosystem assessment, and the quantification of ecosystem carbon storage (<xref ref-type="bibr" rid="B10">Felton et al., 2020</xref>; <xref ref-type="bibr" rid="B16">Hermosilla et al., 2022</xref>; <xref ref-type="bibr" rid="B49">Wang and Gamon, 2019</xref>; <xref ref-type="bibr" rid="B51">Xiao et al., 2019</xref>). Tree species distribution information is vital for various research, such as assessing the impact of extreme weather events or climate change on forest ecosystems, and improving biomass estimation accuracy through tree species data (<xref ref-type="bibr" rid="B9">Fassnacht et al., 2014</xref>; <xref ref-type="bibr" rid="B24">Kumar et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Lindner et al., 2010</xref>; <xref ref-type="bibr" rid="B57">Zhang et al., 2023</xref>). Forest biodiversity is closely tied to productivity and influenced by tree species composition, and biodiversity loss can significantly reduce forests carbon absorption capacity, affecting global carbon sequestration (<xref ref-type="bibr" rid="B43">Shirima et al., 2015</xref>). In many regions, tree species maps are typically derived from field surveys, the high costs of surveys often result in forest inventories being conducted only at multi-year intervals and species maps are not wall-to-wall. This does not meet the needs for mapping forest species over larger areas, and the accessibility and adequacy of existing data remain challenges (<xref ref-type="bibr" rid="B2">Blickensd&#x00F6;rfer et al., 2024</xref>; <xref ref-type="bibr" rid="B50">White et al., 2016</xref>).</p>
<p>Remote sensing offers a promising and more cost-effective alternative to traditional field surveys (<xref ref-type="bibr" rid="B23">Kollert et al., 2021</xref>; <xref ref-type="bibr" rid="B25">Lechner et al., 2020</xref>). Earth observation satellites, such as Landsat and Sentinel, provide continuous images of the earth&#x2019;s surface. These satellites can capture repeated images every few days and compose satellite image time series (SITS), which provide massive information about land cover (<xref ref-type="bibr" rid="B33">Miller et al., 2024</xref>). Moreover, these data can be processed and accessed for free through Google Earth Engine (GEE). Previous research has shown that the use of SITS yields excellent results, especially in vegetation studies, as multi-temporal images capture the phenological characteristics of plants more effectively (<xref ref-type="bibr" rid="B11">Foerster et al., 2012</xref>; <xref ref-type="bibr" rid="B15">Hemmerling et al., 2021</xref>). Phenological differences cause different tree species to be at various growth stages (e.g., greenness rise, leaf on, greenness fall, leaf off), leading to variations in appearance and physiological traits, resulting in distinct spectral reflectance and change patterns. Identifying distinct phenological characteristics and spectral-temporal changes among species has proven to be helpful for vegetation classification (<xref ref-type="bibr" rid="B1">Asner et al., 2008</xref>; <xref ref-type="bibr" rid="B30">Liu et al., 2023</xref>). Simply stacking multi-temporal images may lead to the omission of critical information due to algorithm limitations. Moreover, SITS often contain redundant features, increasing computational time and reducing classifier accuracy, a phenomenon known as the &#x201C;curse of dimensionality&#x201D;. The key challenge is how to mitigate the impact of high-dimensional data while identifying critical periods and optimal features (<xref ref-type="bibr" rid="B3">Camps-Valls et al., 2007</xref>; <xref ref-type="bibr" rid="B17">Hu et al., 2019</xref>; <xref ref-type="bibr" rid="B31">L&#x00F6;w et al., 2013</xref>). Common Machine Learning (ML) models such as Random Forest (RF), Support Vector Machine (SVM), have been widely explored for tree species classification (<xref ref-type="bibr" rid="B13">Fu et al., 2022</xref>; <xref ref-type="bibr" rid="B15">Hemmerling et al., 2021</xref>; <xref ref-type="bibr" rid="B19">Immitzer et al., 2019</xref>; <xref ref-type="bibr" rid="B32">Melnyk et al., 2023</xref>). However, these models process stacked temporal data independently, failing to capture the temporal dependencies inherent in the input (<xref ref-type="bibr" rid="B18">Ienco et al., 2017</xref>; <xref ref-type="bibr" rid="B20">Interdonato et al., 2019</xref>; <xref ref-type="bibr" rid="B37">Pelletier et al., 2019</xref>). Another problem is that traditional models rely heavily on input feature processing, struggling to leverage relationships between multi-temporal data and information redundancy. Consequently, many studies extracted key indicators based on feature selection algorithms or domain knowledge to enhance feature representation among species (<xref ref-type="bibr" rid="B44">Somers and Asner, 2014</xref>; <xref ref-type="bibr" rid="B17">Hu et al., 2019</xref>; <xref ref-type="bibr" rid="B55">You and Dong, 2020</xref>). Even so, feature engineering is time-consuming and challenging, and the resulting information is limited and highly dependent on the algorithms employed (<xref ref-type="bibr" rid="B7">Dou et al., 2021</xref>; <xref ref-type="bibr" rid="B59">Zhong et al., 2019</xref>).</p>
<p>Artificial intelligence (AI) has made significant advancements in the past few years, studies have shown that neural networks can identify and learn time dependencies in sequential data (<xref ref-type="bibr" rid="B40">Ru&#x00DF;wurm and K&#x00F6;rner, 2020</xref>; <xref ref-type="bibr" rid="B59">Zhong et al., 2019</xref>). In contrast to traditional ML, neural network-based deep learning (DL) automatically extract and learn features from vast amounts of data. Convolutional neural networks (CNNs) extract features in the short term through convolution, but may overlook long-term dependencies. Recurrent neural networks (RNNs) such as Long Short-Term Memory (LSTM), capture time dependencies by iteratively updating hidden state, which encodes information from previous time steps. However, this can increase computational complexity and noise, resulting in reduced efficiency (<xref ref-type="bibr" rid="B39">Ru&#x00DF;wurm and Korner, 2017</xref>; <xref ref-type="bibr" rid="B58">Zhao et al., 2022</xref>). Transformer models based solely on the self-attention mechanism process long-term data without recursion, allowing each time step to compute correlations with others in parallel, thus enhancing training efficiency while mitigating noise accumulation and information loss (<xref ref-type="bibr" rid="B48">Vaswani et al., 2017</xref>). It has significantly impacted DL and has been effectively utilized in remote sensing research (<xref ref-type="bibr" rid="B5">Chen et al., 2022</xref>; <xref ref-type="bibr" rid="B27">Li et al., 2022</xref>). The advantages of Transformer in identifying key features and temporal dependencies makes it an excellent choice for tree species classification based on SITS. Given the scarcity of labeled data, a prevalent strategy involves combining pretrain and fine-tune. Pretraining a model on a large dataset allows it to learn general features and representations, followed by fine-tuning the model on a specific task dataset to adapt it for the intended purpose. This transfer of knowledge effectively enhances the model performance and generalization (<xref ref-type="bibr" rid="B21">Jing and Tian, 2021</xref>; <xref ref-type="bibr" rid="B34">Misra and Van Der Maaten, 2020</xref>). Although DL models demonstrate excellent performance, they are often considered as &#x201C;black boxes&#x201D; because of the hard interpretation of their decision-making processes. Explaining the feature learning pipeline can clarify the complex processes of information capture, supporting the reliability of results, and this explanatory process may also provide valuable insights for users (<xref ref-type="bibr" rid="B29">Lipton, 2018</xref>; <xref ref-type="bibr" rid="B41">Samek et al., 2017</xref>; <xref ref-type="bibr" rid="B52">Xu et al., 2021</xref>).</p>
<p>The Forest Inventory Data typically includes both pure and mixed species units, but many studies focus solely on pure stands, which may oversimplify scenarios like mixed stands and limit model generalization. Moreover, using the same source for validation may overestimate mapping accuracy (<xref ref-type="bibr" rid="B2">Blickensd&#x00F6;rfer et al., 2024</xref>; <xref ref-type="bibr" rid="B8">Fassnacht et al., 2016</xref>). Adjusting the dataset to include more samples from diverse forest stands can address the limitations. Pseudo-labeling is a semi-supervised learning technique that generates labels for unlabeled samples through supervised training with labeled samples, enriching the dataset and improving model generalization (<xref ref-type="bibr" rid="B2">Blickensd&#x00F6;rfer et al., 2024</xref>; <xref ref-type="bibr" rid="B46">Tan et al., 2015</xref>; <xref ref-type="bibr" rid="B60">Zhou, 2018</xref>). Combining DL with pseudo-labeling could achieve higher-quality dataset optimization, making it more representative of the study area forest.</p>
<p>The objective of this study is to employ a Transformer-based model to process SITS for tree species identification, while analyzing spectral-temporal data and interpreting the model. The model was pretrained on unlabeled forest pixels to enhance performance, and the dataset was optimized using Pseudo-labeling to include mixed-species scenes. Then the pretrained model was fine-tuned on the dataset, achieving precise forest tree species and tree species biodiversity mapping. Various methods based on statistic and visualization, were utilized to gain a comprehensive understanding of tree species classification using SITS and explore deep learning model. The original SITS data was analyzed from different perspectives, evaluating the contribution of spectral-temporal features, discussing the similarities and differences among tree species as well as the challenges of classification. Meanwhile, the impact of time series data composition on model performance was assessed and the model mechanism was analyzed.</p>
</sec>
<sec id="S2">
<title>2 Materials</title>
<sec id="S2.SS1">
<title>2.1 Study area</title>
<p>The study area (approx. 35&#x00B0; 58&#x2032;&#x2013;37&#x00B0; 2&#x2032; E and 111&#x00B0; 45&#x2032;&#x2013;112&#x00B0; 32&#x2032; N) is located in the Shanxi Province, China. It includes the Huodong National Coal Mining Area and Taiyue Mountain National Forest Park, the largest forest reservation in Shanxi (<xref ref-type="fig" rid="F1">Figure 1</xref>). The Taiyue Mountain and surrounding forests form an important nature reserve with a temperate continental climate, an average annual temperature of 9.2 &#x00B0;C, and 564 mm of precipitation. The study area covers a total of 7,715.17 km<sup>2</sup>, with elevations from 534 to 2,564 meters, and forests occupy 3,861.14 square kilometers, approximately 50% of the region, which belongs to the temperate northern forest zone. The main dominant tree species include: <italic>Larix principis-rupprechtii</italic> (LP), <italic>Pinus tabuliformis</italic> (PT), <italic>Pinus bungeana</italic> (PB), <italic>Platycladus orientalis</italic> (PO), <italic>Quercus wutaishanica</italic> (QW), <italic>Betula</italic> spp. (BA), <italic>Populus</italic> spp. (PS).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Study area and reference data. <bold>(A)</bold> The location of the study area in Shanxi Province, China. <bold>(B)</bold> The elevation information of study area. <bold>(C)</bold> The distribution of tree species reference data (Background: Sentinel-2 image from July 2023, bands R: 4, G: 3, B: 2; World Hillshade from ArcGIS Pro, Esri). Huodong: The Huodong National Coal Mining Area. Tree species: <italic>Larix principis-rupprechtii</italic> (LP), <italic>Pinus tabuliformis</italic> (PT), <italic>Pinus bungeana</italic> (PB), <italic>Platycladus orientalis</italic> (PO), <italic>Quercus wutaishanica</italic> (QW), <italic>Betula</italic> spp. (BA), <italic>Populus</italic> spp. (PS).</p></caption>
<alt-text>Map series displaying an area in Shanxi Province, China. Panel (a) highlights Shanxi, marking the study area, Huodong, and tree cover. Panel (b) shows an elevation map of the Taiyue Mountain and Huodong, with elevations ranging from 534 to 2564 meters. Panel (c) is a satellite image outlining the study area with various land covers or vegetation types marked by different colors.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g001.tif"/>
</fig>
</sec>
<sec id="S2.SS2">
<title>2.2 Remote sensing data</title>
<p>Based on the GEE platform, we used Sentinel-1 Ground Range Detected (GRD) backscatter products and Sentinel-2 MultiSpectral Instrument (MSI) Level-2A products from 2022 to 2023, including cloud removal, monthly median value extraction, and resampling to a spatial resolution of 10 meters, visit <ext-link ext-link-type="uri" xlink:href="https://developers.google.com/earth-engine/datasets/catalog/sentinel">https://developers.google.com/earth-engine/datasets/catalog/sentinel</ext-link> for more details. We calculated commonly used vegetation indices and additional indices recognized in prior researches as valuable for vegetation remote sensing with Sentinel (<xref ref-type="bibr" rid="B12">Frampton et al., 2013</xref>; <xref ref-type="bibr" rid="B26">Li et al., 2014</xref>; <xref ref-type="bibr" rid="B35">Ngo et al., 2023</xref>; <xref ref-type="bibr" rid="B42">Schulz et al., 2024</xref>). Combined indices with the original bands, 33 variables include: B2-B8, B8A, B11-12, VV, VH, NDVI, GNDVI, LSWI, EVI, NDVIre1-3, NDre1-2, Clre, PSRI, MSAVI, MSRre, MTCI, CCCI, S2REP, RVI, VVVHR, NDIVV. Full names and formulas can be found in <xref ref-type="supplementary-material" rid="DS1">Supplementary Table S1</xref>. Temporal profiles of these indices for different tree species are provided in <xref ref-type="supplementary-material" rid="DS1">Supplementary Figures S1, S2</xref>, the plots highlight species-specific spectral, phenological, and structural differences, such as distinct temporal patterns between deciduous and evergreen trees, and backscatter contrasts between coniferous and broadleaf trees (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S3</xref>).</p>
<p>The European Space Agency (ESA) WorldCover 10 m v200 product was used in this study to extract forest areas mask.</p>
</sec>
<sec id="S2.SS3">
<title>2.3 Ground reference dataset</title>
<p>As reference, we used the Forest Inventory data provided by Taiyue Mountain forest administration. The data was collected around 2020 and 2021 through field surveys and high-resolution images, recording forest resource attributes such as dominant and secondary species, stand age and area in polygons. We selected seven main tree species in the study area as target classes, categorizing other species as &#x201C;Others&#x201D; (mostly broadleaf) for the final tree species mapping (<xref ref-type="fig" rid="F1">Figure 1</xref>). Reference points was randomly generated within each polygon, imposing constraints of over 20 meters between points and 30 meters from plot edges, and the number of points was based on plot area, resulting in a reference dataset. To address the simplification of mixed forest scenarios and accuracy overestimation resulting from the use of only pure-species samples, we also selected mixed-species plots to create pseudo-labeling samples, combining them with pure samples for DL model training and evaluation. The specific methods are detailed in section 3.1. We employed stratified sampling to select 60% of the samples from each species as an independent testing dataset to prevent data leakage (<xref ref-type="table" rid="T1">Table 1</xref>). The remaining 40% of samples were used for training and validation with five-fold cross-validation, each fold involved using one subset for validation. This process ensured validation on entirely different samples, aiming to thoroughly evaluate model performance, ensure generalization ability, and adjust model hyperparameters.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>The dataset for tree species mapping.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Short</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Species</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Type</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Samples (pure: pseudo-labeled)</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Training and validation</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Testing</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">LP</td>
<td valign="top" align="left"><italic>Larix principis-rupprechtii</italic></td>
<td valign="top" align="left">Deciduous</td>
<td valign="top" align="left">8,924 (7,754:1,170)</td>
<td valign="top" align="center">3,569</td>
<td valign="top" align="center">5,355</td>
</tr>
<tr>
<td valign="top" align="left">PT</td>
<td valign="top" align="left"><italic>Pinus tabuliformis</italic></td>
<td valign="top" align="left">Evergreen</td>
<td valign="top" align="left">18,155 (11,936:6,219)</td>
<td valign="top" align="center">7,262</td>
<td valign="top" align="center">10,893</td>
</tr>
<tr>
<td valign="top" align="left">PB</td>
<td valign="top" align="left"><italic>Pinus bungeana</italic></td>
<td valign="top" align="left">Evergreen</td>
<td valign="top" align="left">4,666 (4,170:496)</td>
<td valign="top" align="center">1,866</td>
<td valign="top" align="center">2,800</td>
</tr>
<tr>
<td valign="top" align="left">PO</td>
<td valign="top" align="left"><italic>Platycladus orientalis</italic></td>
<td valign="top" align="left">Evergreen</td>
<td valign="top" align="left">8,532 (6,856:1,676)</td>
<td valign="top" align="center">3,412</td>
<td valign="top" align="center">5,120</td>
</tr>
<tr>
<td valign="top" align="left">QW</td>
<td valign="top" align="left"><italic>Quercus wutaishanica</italic></td>
<td valign="top" align="left">Deciduous</td>
<td valign="top" align="left">13,802 (8,710:5,092)</td>
<td valign="top" align="center">5,520</td>
<td valign="top" align="center">8,282</td>
</tr>
<tr>
<td valign="top" align="left">BA</td>
<td valign="top" align="left"><italic>Betula</italic> spp.</td>
<td valign="top" align="left">Deciduous</td>
<td valign="top" align="left">8,053 (6,898:1,155)</td>
<td valign="top" align="center">3,221</td>
<td valign="top" align="center">4,832</td>
</tr>
<tr>
<td valign="top" align="left">PS</td>
<td valign="top" align="left"><italic>Populus</italic> spp.</td>
<td valign="top" align="left">Deciduous</td>
<td valign="top" align="left">1,494 (1,322:172)</td>
<td valign="top" align="center">597</td>
<td valign="top" align="center">897</td>
</tr>
<tr>
<td valign="top" align="left">OT</td>
<td valign="top" align="left"><italic>Others</italic></td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left">2242</td>
<td valign="top" align="center">896</td>
<td valign="top" align="center">1,346</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td/>
<td/>
<td valign="top" align="left">65,868</td>
<td valign="top" align="center">26,343</td>
<td valign="top" align="center">39,525</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn><p>The tree species abbreviations, types, sample quantities (including pure units and pseudo-labeled samples); the number of samples used for training and validation in 5-fold cross-validation and for independent testing.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>Additionally, forest mask was used to generate unlabeled pretraining forest samples by creating grid points at 100 m intervals (10 rows/columns for images), resulting in over 380,000 unlabeled samples.</p>
</sec>
</sec>
<sec id="S3">
<title>3 Methods</title>
<sec id="S3.SS1">
<title>3.1 Tree species dataset optimization</title>
<p>Forest inventory data consists of polygonal plots documenting the dominant and secondary tree species. Previous studies commonly used samples from pure units for model training and accuracy assessment (<xref ref-type="bibr" rid="B19">Immitzer et al., 2019</xref>; <xref ref-type="bibr" rid="B54">Yang et al., 2024</xref>), which may reduce dataset representativeness, lead to lower model performance in identifying mixed-species areas and overestimate mapping accuracy (<xref ref-type="bibr" rid="B2">Blickensd&#x00F6;rfer et al., 2024</xref>). It&#x2019;s easier to distinguish between evergreen and deciduous species due to their greater phenological feature differences. To ensure credibility for the pseudo samples added to the dataset, only mixed units of evergreen and deciduous species were used. Specifically, pure units of target tree species were selected firstly to generate pure samples, then evergreen-deciduous mixed units among the seven target tree species were selected to generate unlabeled samples. The process for generating pseudo-labels for unlabeled samples based on deep learning classification model are as <xref ref-type="fig" rid="F2">Figure 2</xref>. First, the pretrained model was fine-tuned using labeled samples from pure units of two species (one evergreen and one deciduous), to obtain a binary classification model. Next, the unlabeled samples of mixed units were input into the binary classification model, which outputs predicted scores ranging from 0 to 1 for each sample. The class with the highest score was assigned as the pseudo-label, and samples with highest scores above 0.9 were added to the final dataset.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Optimizing the dataset through pseudo-labeled sample generation.</p></caption>
<alt-text>Flowchart depicting a process for classifying tree species. Pure units samples are used in a fine-tuning model for one deciduous and one evergreen classification. Mixed units samples and the fine-tuning model feed into a binary deep learning classification model, which predicts and adds pseudo labels to generate a tree species dataset.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g002.tif"/>
</fig>
</sec>
<sec id="S3.SS2">
<title>3.2 Classification model based on transformer</title>
<sec id="S3.SS2.SSS1">
<title>3.2.1 Transformer</title>
<p>The core of Transformer is the self-attention mechanism, which captures internal relationships between elements at any position in a sequence, excelling at handling long-term dependencies and capturing relevant information (<xref ref-type="bibr" rid="B40">Ru&#x00DF;wurm and K&#x00F6;rner, 2020</xref>). The complexity and temporal correlation of SITS data make Transformers an ideal choice for tree species classification tasks. The original pixels time series data <italic>X</italic> = {<italic>x</italic><sub>1</sub>,<italic>x</italic><sub>2</sub>,&#x2026;,<italic>x</italic><sub><italic>n</italic></sub>} is input, <italic>n</italic> representing the month. The data embedding process uses a linear dense layer to project the original time series observation data into a high-dimensional representation <italic>Linear</italic>(<italic>X</italic>), then the sine and cosine functions of different frequencies are used to generate &#x201C;positional encodings&#x201D;, which have the same dimension as the embeddings and are added to them (<xref ref-type="bibr" rid="B48">Vaswani et al., 2017</xref>):</p>
<disp-formula id="S3.Ex1">
<mml:math id="M1">
<mml:mtable displaystyle="true" rowspacing="0pt">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo rspace="5.8pt" stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mo maxsize="210%" minsize="210%">{</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>/</mml:mo>
<mml:msup>
<mml:mn>10000</mml:mn>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi>model</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:msup>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo rspace="5.8pt">,</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>IF</mml:mi>
</mml:mpadded>
<mml:mpadded width="+3.3pt">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mpadded>
<mml:mpadded width="+3.3pt">
<mml:mi>is</mml:mi>
</mml:mpadded>
<mml:mi>even</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="81.6pt" width="+81.6pt">
<mml:mi>cos</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>/</mml:mo>
<mml:msup>
<mml:mn>10000</mml:mn>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi>model</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">,</mml:mo>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mi>IF</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>is</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>odd</mml:mi>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2002;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex2">
<mml:math id="M2">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="60pt" width="+60pt">
<mml:mi>E</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">+</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;&#x2005;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Where <italic>n</italic> is the time position, <italic>i</italic> is the dimension index, and <italic>d</italic><sub><italic>model</italic></sub> is the embedding dimension. The embedding dimension in this study is <italic>d<sub>model</sub></italic> = 512.</p>
<p>The <italic>Em</italic>(<italic>X</italic>) is projected onto three matrices <italic>Q, K, V</italic> by stacking multiple query vectors (q), key vectors (k), and value vectors (v) using corresponding weight matrices <italic>W<sup>Q</sup></italic>, <italic>W<sup>K</sup></italic>, <italic>W<sup>V</sup></italic>, these matrices are updated during the model training. Then, the scaled dot-product attention function measures similarity by dot-multiplying a query vectors (Q) with a set of key vectors (K), normalizes the result by dividing the dimension of key vector <inline-formula><mml:math id="INEQ17"><mml:msqrt><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:msqrt></mml:math></inline-formula>, and maps it to the weighted time series data <italic>X</italic>&#x2032;:</p>
<disp-formula id="S3.Ex3">
<mml:math id="M3">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:msqrt>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:msqrt>
</mml:mfrac>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2006;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>3</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>To address the limitations of single-head attention in capturing information from complex time series data, Transformer introduces multi-head self-attention, which maps <italic>Q</italic>,<italic>K</italic>,<italic>V</italic> to different feature subspaces using distinct linear layers, executes self-attention in parallel across heads, and concatenates outputs to project into the final hidden representation:</p>
<disp-formula id="S3.Ex4">
<mml:math id="M4">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="27.6pt" width="+27.6pt">
<mml:mi>M</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">&#x2026;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2006;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>4</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Where <italic>head</italic><sub><italic>H</italic></sub> is the self-attention output of each head. The model consists of Transformer blocks, including Multi-Head Attention, Feed-forward networks, residual connections, and layer normalization. By stacking multiple layers, the embedded data is processed by the first layer to produce a hidden representation, which serves as input for the next layer, facilitating information flow and the progressive extraction of high-level feature representations (<xref ref-type="fig" rid="F3">Figure 3A</xref>). This study employed multi-head attention with 8 heads and stacked 3 Transformer layers to capture the complex relationships and information of time series data for identifying tree species.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p><bold>(A)</bold> Transformer model architecture. <bold>(B,C)</bold> Illustrate the pretraining and fine-tuning processes.</p></caption>
<alt-text>Diagram illustrating a machine learning workflow with three sections: (a) Transformer architecture for time series data, displaying layers of blocks leading to hidden representations. (b) Pretraining phase using unlabeled forest samples to develop a pretrained model. (c) Fine-tuning phase with labeled samples for downstream tasks, predicting species with a classification model. Color-coded boxes highlight each process.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g003.tif"/>
</fig>
</sec>
<sec id="S3.SS2.SSS2">
<title>3.2.2 Classification model construction</title>
<p>Self-supervised learning involves pretraining models on specific tasks to learn feature representations from the data, which can be applied to enhance model performance in downstream tasks, especially when the amount of labeled data is limited (<xref ref-type="bibr" rid="B21">Jing and Tian, 2021</xref>). In this study, we referenced the method of <xref ref-type="bibr" rid="B56">Yuan and Lin (2021)</xref> and employed a pretrain task that predicts the temporal data values of samples, enabling the model to learn the spectral-temporal feature context of tree pixels from a large set of unlabeled data (<xref ref-type="fig" rid="F3">Figures 3B,C</xref>). During the pretrain stage, time series data from unlabeled forest samples, which were generated from the forest mask, were used to pretrain model. Specifically, we added uniformly distributed noise between &#x2212;0.5 and 0.5 to the feature values of 4 time points from the 24 in the time series, and trained the model to predict original values of the noisy feature points, using Mean Square Error (MSE) as the optimization function:</p>
<disp-formula id="S3.Ex5">
<mml:math id="M5">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00A0;</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>&#x00A0;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle='true'>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent='true'>
<mml:mi>o</mml:mi>
<mml:mo>&#x00AF;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;&#x2005;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>5</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>n</italic> is the number of time points with added noise, <italic>o<sub>i</sub></italic> is the original value, and<inline-formula><mml:math id="INEQ21"><mml:msub><mml:mpadded lspace="5pt" width="+5pt"><mml:mover accent="true"><mml:mi>o</mml:mi><mml:mo>&#x00AF;</mml:mo></mml:mover></mml:mpadded><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is the predicted value. The model has a hidden size of 512, uses the Adam optimizer with a learning rate of 1e-4, a batch size of 512, is pretrained for 60 epochs (with 10 warm-up epochs), and has a dropout rate of 0.1.</p>
<p>After the pretrain stage, we extracted time series feature values for tree species sample along with their species labels, adapting the pretrained model for class identification by altering the output layer to create a classification model that maps input data to tree species. The model was fine-tuned using labeled sample data, employing Cross-Entropy loss as the optimization function to measure the difference between the model&#x2019;s output and the actual labels for adjusting model parameters:</p>
<disp-formula id="S3.Ex6">
<mml:math id="M6">
<mml:mi>C</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>p</mml:mi>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00A0;</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>&#x00A0;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle='true'>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x00A0;</mml:mo>
<mml:mtext>log</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent='true'>
<mml:mi>y</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
<mml:mo separator="true">&#x2003;&#x2006;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>6</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>C</italic> the number of classes, <italic>y<sub>i</sub></italic> is the ground truth, and <inline-formula><mml:math id="INEQ23"><mml:msub><mml:mover accent="true"><mml:mpadded lspace="5pt" width="+5pt"><mml:mi>y</mml:mi></mml:mpadded><mml:mo>^</mml:mo></mml:mover><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is he predicted probability for class <italic>i</italic>. The model is fine-tuned for 100 epochs using the Adam optimizer with a learning rate of 2e-4 and a batch size of 512.</p>
</sec>
</sec>
<sec id="S3.SS3">
<title>3.3 Spectral-temporal traits analysis</title>
<sec id="S3.SS3.SSS1">
<title>3.3.1 Separability index between tree species</title>
<p>Intra-class and inter-class variability are crucial metrics for evaluating a feature set&#x2019;s ability to distinguish classed, meaning that a class should be most accurately classified when it maximizes differences with other classes while maintaining internal consistency (<xref ref-type="bibr" rid="B17">Hu et al., 2019</xref>). The Separability Index (SI) is used to describe the separability between pairs of classes (<xref ref-type="bibr" rid="B45">Somers et al., 2010</xref>), with higher SI values indicating better separability of the two species in a specific spectral-temporal feature. It is calculated using the following formula:</p>
<disp-formula id="S3.Ex7">
<mml:math id="M7">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="36pt" width="+36pt">
<mml:mi>S</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mn>1.96</mml:mn>
</mml:mpadded>
<mml:mo rspace="5.8pt">&#x00D7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x03C3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x03C3;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>7</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Where <italic>i</italic>,<italic>j</italic> represent two different species, while <italic>M<sub>i</sub></italic>,<italic>M</italic><sub><italic>j</italic></sub> and &#x03C3;<sub><italic>i</italic></sub>,&#x03C3;<sub><italic>j</italic></sub> denote the sample means and variances for the respective species. <italic>v</italic> = {<italic>NDVI</italic>,<italic>GNDVI</italic>,<italic>LSWI</italic>,&#x2026;,<italic>NDIVV</italic>} represents the features used to calculate SI, and <italic>d</italic>={1,2,3,&#x2026;,24} indicates the time points (months).</p>
<p>We calculated the SI for all species pairs to assess key variables and moments for distinguishing tree species. To comprehensively assess the separability among tree species, we averaged the paired indices to obtain a global separability index (SI-global), calculated as follows:</p>
<disp-formula id="S3.Ex8">
<mml:math id="M8">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="36pt" width="+36pt">
<mml:mi>S</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2005;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>8</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Where <italic>SI</italic><sub><italic>ij</italic></sub> represents the separability index between <italic>i</italic> and <italic>j</italic>, <italic>v</italic> and <italic>d</italic> is features and time points. We obtained a matrix with dimensions (<italic>v</italic>,<italic>d</italic>) representing the global separability index for each feature variable at all time points, indicating the overall capability to distinguish between the various tree species.</p>
</sec>
<sec id="S3.SS3.SSS2">
<title>3.3.2 Principal component analysis</title>
<p>Principal component analysis (PCA) is commonly used for dimensionality reduction and serves as a descriptive statistical method to explain variance in multi-dimensional datasets. It projects original multi-dimensional data into a new coordinate system through linear transformations, consolidating information into a few principal components (PCs) while maximizing variance retention and reducing dimensionality (<xref ref-type="bibr" rid="B22">Jolliffe and Cadima, 2016</xref>). We applied PCA on tree species samples&#x2019; bands and index data. The data was standardized, and the covariance matrix was calculated to extract the main independent variables and select the principal components that explain the majority of the variance. We then analyzed the PCs and their relationships with the original features (bands and indices), aiming to assess the significance of these features. Additionally, PCA was employed on each month&#x2019;s data separately to reveal that the importance of individual variables varies throughout the year.</p>
</sec>
</sec>
<sec id="S3.SS4">
<title>3.4 Interpretation for deep learning classification model</title>
<sec id="S3.SS4.SSS1">
<title>3.4.1 Analysis of self-attention weight matrices</title>
<p>Extracting the self-attention weight matrix from the trained model reveals how it focuses on different time points or features, illustrating the influence of low-level inputs on high-level features and the mechanisms of feature transformation (<xref ref-type="bibr" rid="B40">Ru&#x00DF;wurm and K&#x00F6;rner, 2020</xref>). The weight matrix is an <italic>n</italic>-dimensional square matrix (<italic>n</italic> is the sequence length), where each element <italic>weight</italic><sub><italic>ij</italic></sub> indicates the attention of time point <italic>i</italic> on <italic>j</italic>. High weight values suggest the significance of certain time points in relation to high-level features. Transformer employs multi-head self-attention, where each head independently focuses on different parts of the input sequence. This allows the model to capture diverse features by highlighting various relationships, resulting in distinct distributions of high values across weight matrices. Examining the weight matrices of different heads reveals key time points and their relationships, helping to reveal underlying patterns in the data and understand the model&#x2019;s decision-making process. Integrating the weights from all heads provides a global perspective across different layers, revealing overall attention patterns and enhancing our understanding of how the model processes input sequence data. We extracted and visualized the attention weight matrices for each tree species from all heads of layers.</p>
</sec>
<sec id="S3.SS4.SSS2">
<title>3.4.2 Hidden features visualization</title>
<p>In Transformer neural network, input data is processed through multiple layers, resulting in increasingly complex features. This hierarchical structure gradually builds higher-level abstract features from simple raw characteristics. When processing SITS, the model first captures low-level features, such as changes in pixel reflectance and index fluctuations. As the data passes through intermediate layers, the model may identify patterns or trends in the time series, such as specific spectral variations for certain pixel types and differences between categories. In the deeper layers, the model extracts high-level features, which may include complex spatiotemporal relationships and changes in time series values, allowing it to link deep temporal features to the target task and improve accuracy.</p>
<p>We employed t-distributed Stochastic Neighbor Embedding (t-SNE) to visualize the hidden feature outputs from different layers of the classification model. t-SNE projects high-dimensional data points into a lower-dimensional space while preserving the neighborhood relationships of the original data, enabling better visualization in the low-dimensional space (<xref ref-type="bibr" rid="B47">van der Maaten and Hinton, 2008</xref>). The hidden features from different layers was projected into a two-dimensional space to visualize dynamics of the extracted features, approximating how the model extracts and captures key features from multidimensional time series data to distinguish between different tree species during classification.</p>
</sec>
<sec id="S3.SS4.SSS3">
<title>3.4.3 Evaluation of soft outputs</title>
<p>The classification model analyzes the input time series data and applies a Softmax function to produce soft outputs, which are normalized prediction scores for each tree species. These scores indicate the model&#x2019;s estimated probability of the sample belonging to each species, with the highest score determining the predicted output. In addition to evaluating the performance of the model based solely on overall or class-specific accuracy, we tried to examine how different constructions of time series data impact predictions from an &#x201C;internal&#x201D; perspective. The predicted scores of test samples&#x2019; true classes were used as a reference metric, which indirectly reflects the confidence and reliability of the classification model outputs. By observing how these predicted scores vary with changes in the time series data, we aim to dynamically monitor how the input data composition influences multi-species classification and the resulting accuracy variations.</p>
</sec>
</sec>
<sec id="S3.SS5">
<title>3.5 Biodiversity estimation</title>
<p>To further analyze the forest ecosystem in the study area, we calculated biodiversity indices based on the final tree species classification map: Species Richness, the Shannon-Wiener Index, and Simpson&#x2019;s Diversity Index (<xref ref-type="bibr" rid="B38">Peng et al., 2021</xref>). Species Richness reflects diversity by simply counting the number of species in a sample. The Shannon-Wiener Index takes into account both species richness and evenness, based on the relative abundance of each species in the sample. Simpson&#x2019;s Diversity Index emphasizes dominant species by considering the square of relative abundance, highlighting species evenness. The formulas are as follows:</p>
<disp-formula id="S3.Ex9">
<mml:math id="M9">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mpadded lspace="60pt" width="+63.3pt">
<mml:mi>R</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;&#x2006;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>9</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex10">
<mml:math id="M10">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mpadded lspace="60pt" width="+63.3pt">
<mml:mi>H</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x03A3;</mml:mi>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mi>i</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2006;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>10</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex11">
<mml:math id="M11">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mpadded lspace="60pt" width="+63.3pt">
<mml:mi>D</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x03A3;</mml:mi>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mi>i</mml:mi>
</mml:mpadded>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
<mml:mo>&#x2062;</mml:mo>
<mml:mmultiscripts>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:none/>
<mml:none/>
<mml:mn>2</mml:mn>
</mml:mmultiscripts>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>11</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Where <italic>R</italic> is Species Richness, <italic>H</italic> is Shannon-Wiener Index, <italic>D</italic> is Simpson&#x2019; s Diversity Index, <italic>S</italic> is total number of species in sample, and <italic>p<sub>i</sub></italic> is the proportion of individuals of species <italic>i</italic> relative to the total number of individuals. we calculated each index at a resolution of 100 m.</p>
</sec>
<sec id="S3.SS6">
<title>3.6 Accuracy evaluation</title>
<p>We generated a confusion matrix from validation results and calculated six commonly used metrics: Overall Accuracy (OA), Kappa, User&#x2019;s Accuracy (UA), Producer&#x2019;s Accuracy (PA), F1 Score and macro-F1. UA (Precision) and PA (Recall) focus on class-specific accuracy, while the F1 Score combines both for a comprehensive measure. OA, Kappa and macro-F1 evaluate overall performance.</p>
<disp-formula id="S3.Ex12">
<mml:math id="M12">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="60pt" width="+60pt">
<mml:mi>U</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>A</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2002;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>12</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex13">
<mml:math id="M13">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="60pt" width="+60pt">
<mml:mi>P</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>A</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>13</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex14">
<mml:math id="M14">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="60pt" width="+60pt">
<mml:mi>O</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>A</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>14</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex15">
<mml:math id="M15">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="36pt" width="+36pt">
<mml:mi>K</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>a</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>-</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>15</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex16">
<mml:math id="M16">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mpadded lspace="12pt" width="+12pt">
<mml:mi>F</mml:mi>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>e</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:mn>2</mml:mn>
</mml:mpadded>
<mml:mo rspace="5.8pt">&#x00D7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>A</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">&#x00D7;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>16</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="S3.Ex17">
<mml:math id="M17">
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x00A0;</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>&#x00A0;</mml:mo>
<mml:mstyle displaystyle='true'>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mstyle>
<mml:mo>/</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo separator="true">&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>17</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>k</italic> is the number of classes, <italic>TP</italic> is the number of true positives, <italic>FP</italic> is the number of false positives, <italic>FN</italic> is the number of false negatives,<italic>N<sub>correct</sub></italic> is the number of corrected classified samples, <italic>N</italic><sub><italic>total</italic></sub> is the number of all samples, <italic>P<sub>o</sub></italic> is the overall accuracy, <italic>P<sub>e</sub></italic> is the proportion of agreement expected by chance.</p>
<p>Considering the potential bias caused by the spatial distribution of samples, we adopted an inverse sampling-intensity weighted method (<xref ref-type="bibr" rid="B6">De Bruin et al., 2022</xref>). This method corrects estimation bias by assigning more weight to sparsely sampled areas and less weight to densely sampled areas, based on the sample distribution density. Using the Kernel Density Estimation (KDE) method from the scikit-learn package in Python (<xref ref-type="bibr" rid="B36">Pedregosa et al., 2011</xref>), we estimated the sample point density and applied inverse sampling weights. Additional accuracy metrics, including weighted-OA, weighted-kappa, weighted-F1 and weighted-macro-F1, were calculated to provide a more comprehensive evaluation of accuracy.</p>
</sec>
</sec>
<sec id="S4" sec-type="results">
<title>4 Results</title>
<sec id="S4.SS1">
<title>4.1 Model validation</title>
<p>Five-fold cross-validation was employed to evaluate the model performance and generalization ability based on 2 years data, with accuracy variations and F1-scores for each species shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. The model showed stable performance across different folds, with an average Kappa of 0.78, OA of 0.82, and macro-F1 of 0.80. The F1-scores for each tree species were also consistent, with average values as follows: LP 0.83, PT 0.88, PB 0.76, PO 0.83, QW 0.76, PA 0.84, PS 0.76, OT 0.75. The lowest standard deviation was for PT 0.003, while the LP, PB, PO, QW, and BA were around 0.013, highest standard deviation was for OT 0.037. These results indicate that the model is stable and reliable for tree species classification, and the final accuracy to be confirmed by the testing samples.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>The results of 5-fold cross-validation. <bold>(A)</bold> Box plots for Kappa, OA, and macro-F1. <bold>(B)</bold> F1 scores for each tree species (bars represent the mean values, and error bars indicate the standard deviation).</p></caption>
<alt-text>Box plots and bar charts illustrating statistical scores. The top graph (a) shows box plots comparing values for kappa, OA, and macro-F1, ranging from 0.70 to 0.85. The bottom graph (b) presents F1-scores using bar charts for categories LP, PT, PB, PO, QW, BA, PS, and OT, with values around 0.6 to 0.8.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g004.tif"/>
</fig>
</sec>
<sec id="S4.SS2">
<title>4.2 The impacts of time series data construction</title>
<p>We first tested the model using only data from June to July 2022, and the results achieved an OA of only 0.49, Kappa of 0.38 and macro-F1 of 0.38. We then gradually extended the time series by adding 1 month of data at both the start and end in each iteration, analyzing the impact of varying time series length (<xref ref-type="fig" rid="F5">Figure 5</xref>). The results showed that increasing the input length gradually improves the performance of the model. When the series was extended to include data from April to September, which is commonly used for vegetation and crop remote sensing studies, the classification accuracy significantly improved, with an OA of 0.68, Kappa of 0.62, and macro-F1 of 0.64. This time range covers the greenness rise and fall period for most tree species, as well as the leaf-on and off period for deciduous species. As the temporal depth increased to cover the late leaf-fall period of deciduous trees, feature differences between evergreen and deciduous species grew, resulting in a gradual improvement in each classification accuracy metrics. The OA approached 0.8, with a macro-F1 of 0.78, using monthly data for all of 2022, demonstrating effective species differentiation.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Accuracy and F1 Score variations of different time series construction (1&#x2013;12 represent January to December 2022, 13&#x2013;24 represent January to December 2023).</p></caption>
<alt-text>Line graphs showing changes in accuracy metrics and F1 scores over specified month ranges. The left graph shows OA, Kappa, and Macro-F1 metrics, all increasing and stabilizing around 0.8 to 0.9 over time. The right graph displays F1 scores for multiple species, each line showing a similar upward trend, with scores reaching around 0.8 to 0.9.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g005.tif"/>
</fig>
<p>Additionally, we incorporated data from the following year (2023) into the time series, adding one month at a time (13&#x2013;24 months represent January to December 2023). This approach aimed to determine whether extending the time series beyond an entire growth cycle would enhance accuracy of the model and capture more potential time dependencies and hidden features across annual data. The results showed that extending the length slightly improved classification accuracy, but the gains were much smaller than those seen when the temporal data did not cover a complete growth cycle. Moreover, simply extending the time series does not guarantee improved accuracy. For instance, when the data was extended by an additional 3 months (i.e., 1&#x2013;15 months) beyond the full year of 2022, here was little improvement in accuracy, with the macro-F1 fluctuating around 0.77 and no increase in OA or Kappa. This indicating that the new data may lack significant additional information. As data from the second year (2023) growing season was incorporated, all metrics continued to grow gradually. Extending the temporal length to 24-months, covering 2 years, all accuracy metrics reached their highest level: OA 0.847, Kappa 0.815, and macro-F1 0.836, Both the accuracy metrics and the F1 score growth curve approached a stable state (<xref ref-type="fig" rid="F5">Figure 5</xref>). We also tested the model using only Sentinel-2 24-months data for classification, and the results showed slightly lower accuracy compared to the combination of two data sources, with an OA of 0.819, Kappa of 0.782, and macro-F1 of 0.805 (confusion matrix in <xref ref-type="supplementary-material" rid="DS1">Supplementary Table S2</xref>).</p>
<p>The Inverse sampling-intensity weighted method was used to account for potential estimation bias from the spatial distribution of samples. Additional weighted accuracy metrics were calculated, with the confusion matrix shown in <xref ref-type="table" rid="T2">Table 2</xref> (Estimated sampling intensity and sample weights distribution in <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S4</xref>). The results show a slight decrease in F1 scores for several species, with PO and BA decreasing from 0.85 and 0.86 to 0.81, respectively. There were also adjustments in the overall accuracy metrics, with OA decreasing from 0.847 to 0.834 and macro-F1 dropping from 0.836 to 0.813. These adjustments further improved the reliability of the results, and the overall performance remains satisfactory.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Confusion matrix of the classification model using 24-months data.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;" rowspan="2">Map class</td>
<td valign="top" align="center" colspan="8" style="color:#ffffff;background-color: #7f8080;">Reference class (samples)</td>
<td valign="top" align="center" colspan="4" style="color:#ffffff;background-color: #7f8080;">Accuracy</td>
</tr>
<tr>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">LP</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">PT</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">PB</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">PO</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">QW</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">BA</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">PS</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">OT</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">UA</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">PA</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">F1 score</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Weighted-F1 score</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">LP</td>
<td valign="top" align="center">4,349</td>
<td valign="top" align="center">229</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">117</td>
<td valign="top" align="center">181</td>
<td valign="top" align="center">15</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="top" align="left">PT</td>
<td valign="top" align="center">253</td>
<td valign="top" align="center">9,411</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">43</td>
<td valign="top" align="center">276</td>
<td valign="top" align="center">113</td>
<td valign="top" align="center">74</td>
<td valign="top" align="center">21</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.89</td>
</tr>
<tr>
<td valign="top" align="left">PB</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">46</td>
<td valign="top" align="center">2,318</td>
<td valign="top" align="center">277</td>
<td valign="top" align="center">357</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">33</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.77</td>
</tr>
<tr>
<td valign="top" align="left">PO</td>
<td valign="top" align="center">13</td>
<td valign="top" align="center">83</td>
<td valign="top" align="center">270</td>
<td valign="top" align="center">4,472</td>
<td valign="top" align="center">423</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">21</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="top" align="left">QW</td>
<td valign="top" align="center">230</td>
<td valign="top" align="center">793</td>
<td valign="top" align="center">177</td>
<td valign="top" align="center">293</td>
<td valign="top" align="center">6,763</td>
<td valign="top" align="center">212</td>
<td valign="top" align="center">29</td>
<td valign="top" align="center">124</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.79</td>
</tr>
<tr>
<td valign="top" align="left">BA</td>
<td valign="top" align="center">427</td>
<td valign="top" align="center">181</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">157</td>
<td valign="top" align="center">4,279</td>
<td valign="top" align="center">23</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="top" align="left">PS</td>
<td valign="top" align="center">48</td>
<td valign="top" align="center">93</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">756</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.78</td>
</tr>
<tr>
<td valign="top" align="left">OT</td>
<td valign="top" align="center">28</td>
<td valign="top" align="center">57</td>
<td valign="top" align="center">28</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">167</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">1,126</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="top" align="right" colspan="9">OA = 0.847</td>
<td valign="top" align="right" colspan="2">Kappa = 0.815</td>
<td valign="top" align="right" colspan="2">macro-F1 = 0.836</td>
</tr>
<tr>
<td valign="top" align="right" colspan="9">OA_w = 0.834</td>
<td valign="top" align="right" colspan="2">Kappa_w = 0.791</td>
<td valign="top" align="right" colspan="2">macro-F1_w = 0.813</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn><p>Confusion matrix along with accuracy metrics, and adjusted metrics using Inverse sampling-intensity weighted (Weighted-F1 Score, OA_w, Kappa_w, and macro-F1_w represent the weighted metrics).</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="S4.SS3">
<title>4.3 Forest tree species and biodiversity map</title>
<p>By using 24 months data, the model demonstrated its ability to produce high-quality tree species classification. We applied the trained model to process time series data of all forest pixels, producing a tree species distribution map for the entire region (<xref ref-type="fig" rid="F6">Figure 6</xref>). The map showed good spatial consistency with the forest inventory reference data, demonstrating the reliability of the classification outcomes. The area statistics for each tree species are as follows: <italic>Pinus tabuliformis</italic> covers the largest area at 2,368.54 km<sup>2</sup>, followed by <italic>Quercus wutaishanica</italic> at 760.14 km<sup>2</sup>. <italic>Larix principis-rupprechtii</italic>, <italic>Pinus bungeana</italic> and <italic>Betula</italic> spp. have similar coverage, with 138.64 km<sup>2</sup>, 145.14 km<sup>2</sup>, and 136.29 km<sup>2</sup>, respectively. <italic>Populus</italic> spp. occupies a small area of 17.36 km<sup>2</sup>, while <italic>Others</italic> (mainly broadleaf) cover 123.53 km<sup>2</sup>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Tree species map of study area. <bold>A-G</bold> represent the comparison regions with: (1) Sentinel-2 RGB from July 2023 (2) forest investigation map and (3) predicted tree species map.</p></caption>
<alt-text>Map showing tree species distribution over a defined area with labeled sections A to G. Each section has corresponding satellite images showing terrain and vegetation. The legend identifies different tree species with colors: LP (yellow), PB (blue), QW (green), PS (light blue), PT (maroon), PO (orange), BA (dark blue), OT (beige). A scale bar and compass indicate orientation and distance. Sections A to G are detailed with smaller comparative images, highlighting regional vegetation and landscape features.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g006.tif"/>
</fig>
<p>Three biodiversity indices were calculated using the tree species distribution map and displayed the results (<xref ref-type="fig" rid="F7">Figure 7</xref>). The Species Richness ranges from 1 to 8, indicating the number of tree species within a unit. The Shannon-Wiener Index ranges from 0 to 1.949, with higher values indicating a more complex and even tree species composition. The Simpson&#x2019;s Diversity Index ranges from 0 to 0.851, with values closer to 1 indicating that most individuals are concentrated in a few species, reflecting lower evenness. These results illustrate the biodiversity of the study area forest, offering insights into the health and resilience of ecosystem.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Biodiversity indices of the forest in study area. <bold>(A)</bold> Species Richness. <bold>(B)</bold> Shannon-Wiener Index. <bold>(C)</bold> Simpson&#x2019;s Diversity Index.</p></caption>
<alt-text>Three maps show biodiversity metrics for Huodong. Map (a) depicts species richness with values from 1 to 8. Map (b) displays the Shannon-Wiener index ranging from 0 to 1.949. Map (c) illustrates the Simpson index from 0 to 0.851. Each map uses a similar color gradient to represent values within the outlined Huodong region.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g007.tif"/>
</fig>
</sec>
<sec id="S4.SS4">
<title>4.4 Key features analysis by statistical methods</title>
<p>The global separability index (SI-global) combines the feature variable separability of various tree species, as shown in <xref ref-type="fig" rid="F8">Figure 8A</xref>, where the intensity of colors indicates differentiation of a feature variable among tree species at a specific time point, reflecting its overall capability to distinguish between species. The results showed that the vegetation indices (e.g., NDVI, GNDVI, NDre1, CIre, etc.) calculated from NIR and red-edge bands demonstrate higher separability, compared to the original band values from satellite sensors. The specific separability indices of all pairs are shown in <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S5</xref>. We found that the distinction between deciduous and evergreen species exhibited significantly separability, and extracted the separability index map (<xref ref-type="fig" rid="F8">Figures 8B,C</xref>) for four pairs of tree species (one pair of evergreen species (PT-PO), one pair of deciduous species (LP-BA), and two pairs of mixed deciduous and evergreen species (PT-BA and LP-PT).) The SI heatmaps for same-type species pairs showed low separability, indicating minimal differences in shallow features. In contrast, different species pairs exhibited significant separability, particularly from January to April, October to December, and July to September, with higher index values during these periods. Most deciduous trees are in leafless stages during these periods, leading to significant differences from evergreens in spectral reflectance. These results also highlighted the challenge of distinguishing species with similar phenology.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption><p>Separability index. The horizontal and vertical axes represent time points (months) and SI of variables, and the color of each cell indicates the separability value. <bold>(A)</bold> SI-global: Overall separability across all tree species pairs for spectral-temporal features. <bold>(B)</bold> SI for two pairs of the same tree type (both evergreen or both deciduous). <bold>(C)</bold> SI for two pairs of different tree types (evergreen vs. deciduous).</p></caption>
<alt-text>Heatmaps display different vegetation indices over a 24-month period. Panel (a) shows global indices, while panels (b) and (c) compare various site combinations: PT vs PO, LP vs BA, PT vs BA, and LP vs PT. Shades range from light to dark blue, indicating varying index values. Each panel includes a color scale for reference.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g008.tif"/>
</fig>
<p>PCA was used for dimensionality reduction on the dataset with 33 variables. The results (<xref ref-type="fig" rid="F9">Figure 9A</xref>) showed that the majority of the variance in tree species samples is concentrated in PC1 (37.91%) and PC2 (19.83%). We further analyzed the composition of the two principal components by displaying the distribution of variables (<xref ref-type="fig" rid="F9">Figure 9B</xref>) and the top 10 important variables for each component (<xref ref-type="fig" rid="F9">Figures 9C,D</xref>). Vegetation indices have high contributions in PC1, while optical bands dominate PC2, and SAR data shows minimal contribution in the PCA analysis. In PC1, the red edge indices (NDre1, NDre2, MSReN, MSRre, NDVIre1, Clre) and the optical vegetation indices calculated using near-infrared bands (NDVI, mSAVI, GNDVI, LSWI) contribute significantly, these variables primarily reflect the physiological indicators such as chlorophyll. PC2 is primarily influenced by the original optical satellite bands, reflecting the surface forest canopy&#x2019;s spectral information, such as leaf color and brightness. To further investigate the importance of each variable at different time points, PCA was performed on monthly data (2023) to assess the contributions to the first two PCs, ranking changes are shown in <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S6</xref>, and monthly variable loadings in <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S7</xref>. While contributions fluctuated over time, the trend remained consistent with previous analysis. Vegetation indices made the largest contribution to PC1, consistently showing the highest contributions across different time. In contrast, PC2 was primarily influenced by the Sentinel-2 bands.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption><p>Principal component analysis. <bold>(A)</bold> Explained variance of first 8 principal components. <bold>(B)</bold> Biplot of Variable Loadings on PC1 and PC2. <bold>(C, D)</bold> Contributions of top 10 variables to PC1 and PC2.</p></caption>
<alt-text>(a) Bar chart showing explained variance by principal components, with PC1 accounting for 37.91% and PC2 for 19.83%. (b) Biplot of variables&#x2019; contributions to principal components. (c) Bar chart of variables&#x2019; contributions to Principal Component 1, with MDE1 and MDE2 having the highest contributions. (d) Bar chart of variables&#x2019; contributions to Principal Component 2, with B5 and B3 contributing most.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g009.tif"/>
</fig>
</sec>
<sec id="S4.SS5">
<title>4.5 Multidimensional analysis of deep learning classification model</title>
<sec id="S4.SS5.SSS1">
<title>4.5.1 Self-attention weight distribution</title>
<p>The multi-head and layers architecture of the model helps capture information from complex time series data, focusing on different aspects and learning dependencies at multiple levels. In the first layer, weight matrices across tree species are similar, but differences emerge in the second and third layers. We used one species (LP) as an example to illustrate how the model processes information, and all attention weights matrices from the first layer for LP samples was visualized firstly (<xref ref-type="fig" rid="F10">Figure 10A</xref>). In the initial layer, the model aims to extract as much useful information as possible from the input by focusing on various time points. The attention weight matrices showed dispersed attention to help capture global features and some weight matrices exhibited patterns resembling spectral characteristics of tree. For instance, head-1 exhibited high self-attention occurs between 5&#x2013;9 and 17&#x2013;21 months (5&#x2013;9 months of 2023), likely capturing seasonal features linked to leaf growth to shedding periods, as well as the fluctuation of vegetation indices. Conversely, head-3 corresponded to the leafless period of deciduous trees, highlighting the differentiation from evergreens. Head-6 and 7 exhibited attention focused on similar time periods, such as 1&#x2013;4 and 10&#x2013;15 or 21&#x2013;24 months (non-growth seasons), as well as between 4 and 9 months and the corresponding period in the following year. These suggested the model capability to capture correlations, demonstrating its effectiveness in extracting cross-year and seasonal information.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption><p>Self-attention weight matrices. The axes represent time points(months). <bold>(A)</bold> Attention weight distributions for the 8 heads in Layer 1 (LP samples). <bold>(B)</bold> Average attention weights across three layers.</p></caption>
<alt-text>Self-attention weight matrices for Layer 1 are shown in eight heatmaps, each representing different attention heads, marked Head 1 to Head 8. Below, average weight matrices for Layers 1, 2, and 3 are presented in three separate heatmaps. The heatmaps use a color scale from dark purple to bright yellow, indicating weight intensity. Axes range from 1 to 24.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g010.tif"/>
</fig>
<p>We averaged the attention weights from all heads in each of the three layers and visualized the resulting matrices to gain a clearer understanding of the model processing mechanism (<xref ref-type="fig" rid="F10">Figure 10B</xref>). In the higher attention layers, the model concentrates on key features and significant time points derived from the shallow features of lower layers, assigning them greater weight. In the second layer, feature extraction converged, with high attention concentrated on fewer time points. The attention weights became more dispersed in next layer, as the model integrates information and revisits valuable features and time points that may have been overlooked. This improvement helps prevent model overfitting to specific features and enhances its generalization, enabling it to more effectively capture both similarities and differences between species, ultimately leading to more reliable identification of tree species.</p>
</sec>
<sec id="S4.SS5.SSS2">
<title>4.5.2 Dynamics of hidden features separability</title>
<p>400 samples for each specie were selected randomly to illustrate the hidden features extraction process. First, samples with the original 33 features was projected into a two-dimensional space by t-SNE (<xref ref-type="fig" rid="F11">Figure 11A</xref>). The results revealed that the samples from different tree species could not be effectively distinguished, highlighting substantial similarities among the species in the original feature space. This indicated that relying solely on the original features makes it difficult to capture and express the distinct differences between tree species.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption><p>Visualization of feature separability based on t-SNE. <bold>(A)</bold> Original features. <bold>(B&#x2013;D)</bold> Hidden features from each layer.</p></caption>
<alt-text>Four scatter plots labeled (a) to (d) show data using t-SNE dimensions with various color-coded clusters. Next to them is a neural network diagram illustrating time series data flow through three layers to an output. The legend defines color labels such as LP, PT, PB, and others.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g011.tif"/>
</fig>
<p>Next, the samples data was inputted into the trained classification model, which consists of three Transformer layers with hidden features of 512 dimensions. The hidden representation outputs from each layer was projected and visualized in a two-dimensional space (<xref ref-type="fig" rid="F11">Figure 11</xref>). The hidden feature output of each layer consists of 512-dimensional representations for each sample across all time points. We averaged these features along the time dimension for t-SNE processing and projection into two-dimensional space. As shown in <xref ref-type="fig" rid="F11">Figure 11</xref>, the hidden features extracted and processed through the Transformer layers exhibited more pronounced clustering of similar samples as the model progresses into deeper layers. After being processed by the first layer, a noticeable trend of clustering among similar samples emerged compared to the projection of original features, with LP and PT samples beginning to form clusters. By the second layer, distinct clustering and separation of all tree species became apparent (<xref ref-type="fig" rid="F11">Figure 11C</xref>), and the projection of final layer showed even more pronounced inter-species separation (<xref ref-type="fig" rid="F11">Figure 11D</xref>). The result showed that the samples of deciduous trees, including LP, QW, BA, PS, are closely positioned after dimensionality reduction. Similarly, the evergreen species PB and PO are also near each other.</p>
</sec>
<sec id="S4.SS5.SSS3">
<title>4.5.3 Soft outputs variations in time series</title>
<p>We tested the impact of time series construction on the classification model in section 4.2, that longer time series data provides additional information beneficial for classification. Additionally, we extracted the soft outputs from models to illustrate how the model confidence in its predictions varies with the increasing length of time series, represented by prediction scores ranging from 0 to 1. The soft outputs for most time series constructions are presented as histograms, divided into 10 intervals of 0.1 (<xref ref-type="fig" rid="F12">Figure 12A</xref>). Details of all tree species prediction scores across these time series constructions are presented in <xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S8</xref>.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption><p>Soft outputs predicted scores. <bold>(A)</bold> The variation in average predicted scores across different time series compositions (Each value represents the sample percentage within each interval, the red dashed lines represent average score for all species). <bold>(B)</bold> The comparison of scores between 12-month (blue) and 24-month (red) data.</p></caption>
<alt-text>Histograms showing average predicted score changes across various months and comparisons of predicted scores for twelve versus twenty-four months. Part (a) depicts score changes from months 6-7 to 1-24, with average scores highlighted by red dashed lines. Part (b) compares scores for twelve and twenty-four months across different categories (OT, LP, PT, PB, PO, QW, BA, PS), with scores indicated by red and blue dashed lines respectively. Each histogram&#x2019;s y-axis represents percentage, while the x-axis shows average scores.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g012.tif"/>
</fig>
<p>Similarly, when using only June-July 2022 data, the lack of temporal variation and significant phenological changes resulted in very low prediction scores, indicating inadequate model confidence and failure to identify tree species, with a macro-F1 below 0.4. With data from May to August, a noticeable shift in sample scores occurred, transitioning from a concentration in the lower range to a more even distribution across each interval. The average score also increased, and the proportion of samples with scores above 0.9 grew significantly. When data was extended to cover April to September, the average score reached 0.63, accompanied by a marked increase in high-scoring samples. The soft output prediction scores revealed the rapid accuracy improvement noted in section 4.2 from an internal model perspective, showing that the richer phenological information from April to September greatly enhances the model confidence in identifying tree species. As the time series covered the entire year of 2022, the proportion of samples with prediction scores between 0.9 and 1.0 rose to 64%, with an average score of 0.8, reflecting improved confidence in the classification model outputs.</p>
<p>Incorporating data from the following year into the time series did not significantly raise the average prediction score, but the proportion of the 0.9&#x2013;1.0 range steadily increased, reaching 70% when covering 2 years, representing a 6% improvement compared to the 1-year results. We compared the prediction scores of each tree species using 1 year (1&#x2013;12) and 2 years (1&#x2013;24) of data (<xref ref-type="fig" rid="F13">Figure 13B</xref>), revealing that extending the time series consistently improved the model prediction confidence for nearly all species. However, the average prediction score and the proportion of samples above 0.9 for PB decreased. Despite this, the F1-scores in the two composition were 0.74 and 0.79, confirming improved classification accuracy. The model likely overestimated PB in the 1 year of data, with an average score of 0.9, significantly higher than other species. The extended time series provided additional information, enabling better feature capture, which optimized the scores while improved classification accuracy.</p>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption><p>Variation in OA, macro-F1 and F1 Score between Non-pretrained and Pretrained models.</p></caption>
<alt-text>Two graphs display performance data. The left line graph shows model performance over 100 epochs, comparing Overall Accuracy (OA) and macro-F1 scores for pretrained and non-pretrained models. Pretrained models perform better. The right bar chart compares F1 scores across different tree species, showing higher scores for pretrained models for each species (LP, PT, PB, PO, QW, BA, PS, OT).</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="ffgc-08-1599510-g013.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="S5" sec-type="discussion">
<title>5 Discussion</title>
<sec id="S5.SS1">
<title>5.1 Pretraining effects on model performance</title>
<p>The pretrain and fine-tune approach was employed to enhance the classification model ability to identify tree species. We trained a model without pretraining, and compared the changes in OA and macro-F1 during the training process with pretrained model, as well as the final F1 scores for each tree species (<xref ref-type="fig" rid="F13">Figure 13</xref>). The results demonstrated that the model, which was pretrained on forest pixels, significantly improved the accuracy for classification, with noticeable enhancements in the identification accuracy across all species. Additionally, the accuracy curve over training epochs shows that the pretrained model reached the performance ceiling of the non-pretrained model around the 20th epoch, further boosting the accuracy ceiling. The cross-validation results further confirm the enhanced consistency of the pretrained model (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S9</xref>). Existing studies have shown that pretraining models can effectively enhance generalization ability and performance (<xref ref-type="bibr" rid="B21">Jing and Tian, 2021</xref>; <xref ref-type="bibr" rid="B34">Misra and Van Der Maaten, 2020</xref>). But creating a pretraining task that aligns well with the target task is challenging and requires further research, such as transfer learning strategies and general models (<xref ref-type="bibr" rid="B33">Miller et al., 2024</xref>).</p>
</sec>
<sec id="S5.SS2">
<title>5.2 Significance of spectral-temporal features</title>
<p>Extending time series data generally improved classification accuracy for tree species in most cases (<xref ref-type="fig" rid="F5">Figure 5</xref>), highlighting the advantage of using time series images with more varying information (<xref ref-type="bibr" rid="B11">Foerster et al., 2012</xref>). Using only June and July data resulted in an OA of just 0.49. Despite the period being the peak growing season for most plants, the shorter series lacked phenological variation, leaving the model to rely almost entirely on spectral reflectance values. Trees with similar phenological traits tend to exhibit highly similar spectral signatures, but key phenological changes are crucial for species identification, and longer time series capturing these shifts provide more information for better differentiation (<xref ref-type="bibr" rid="B30">Liu et al., 2023</xref>). Extending the data beyond a full year can still slightly improve accuracy of identifying tree species, but the gains were less significant than adding within-year data, and some additional data may not contribute valuable information. Visualizing the model soft outputs showed that adding an extra year of data boosts prediction confidence, allowing better identification of subtle differences between tree species and optimizing predictions (<xref ref-type="fig" rid="F12">Figure 12</xref>). As time series length increases, classification accuracy for individual tree species and overall performance stabilized at a satisfactory level, and we believe that further extension may not significantly enhance true classification accuracy.</p>
<p>The physiological differences between deciduous and evergreen trees result in distinct spectral reflectance patterns, leading to a more pronounced separability between these two groups. The results indicated that during the leaf fall period, distinct types tree species show significant separability (<xref ref-type="fig" rid="F8">Figure 8</xref>). In contrast, similar growth processes among the same type species lead to insignificant difference in feature values across time series, making accurate identification difficult using only original bands and indices. Many studies emphasize the significance of NIR and Red-edge indices for vegetation research (<xref ref-type="bibr" rid="B12">Frampton et al., 2013</xref>; <xref ref-type="bibr" rid="B26">Li et al., 2014</xref>), we also found that original bands are less effective for distinguishing between species compared to index variables, even when comparing two distinct types of trees. Vegetation indices effectively highlight differences between tree species by sensitively reflecting physiological indicators like chlorophyll and water content. PCA results also confirm the importance of vegetation index variables (<xref ref-type="fig" rid="F9">Figure 9</xref>), with a high contribution from vegetation indices in the most significant principal component (PC1). PC1 primarily captures seasonal subtle variations in tree physiological characteristics, while PC2, influenced mainly by original bands, reflects differences in leaf color, consistent with previous studies (<xref ref-type="bibr" rid="B42">Schulz et al., 2024</xref>). There is variability in the importance of different variables to the principal components across different time periods, due to differences in sensitivity to phenological characteristics at various stages. However, the primary contributors to the two principal components remain the vegetation indices and spectral bands, respectively. We also calculated Pearson&#x2019;s correlation coefficients among the 33 feature variables to assess inter-variable relationships (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S10</xref>). The results revealed higher correlations and possible redundancy among optical bands with similar wavelengths and their derived vegetation indices, as well as among SAR-derived variables. In contrast, the correlations between variables from the two different sources were generally low. This suggests that adding SAR data to optical inputs may provide complementary information for distinguishing tree species, as the combined data achieved higher tree species classification accuracy compared to using only optical features, with macro-F1 scores of 0.836 and 0.805, respectively. This aligns with previous studies integrating multiple data sources, suggesting that additional information provided by SAR, such as vegetation structural characteristics, may contribute to improved classification performance, which is also reflected in the distinct backscatter profiles across species (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S2</xref>).</p>
</sec>
<sec id="S5.SS3">
<title>5.3 Insights from model visualization</title>
<p>In addition to producing high-accuracy tree species classification maps using deep learning models on SITS, visualizing the model can help us understand its internal processes and bolster its reliability. Multi-head attention weight matrices revealed the model behavior in capturing information from various aspects of the input time series, including key phenological changes and periods of features similarity (section 4.5.1). The model employed nonlinear transformations and multi-head self-attention mechanisms to continuously extract and integrate features, enhancing class-relevant information while diminishing irrelevant features. This process evolves the input features into a more abstract and separable form, illustrating the significant advantage of deep learning in processing time series data and capturing information (<xref ref-type="bibr" rid="B48">Vaswani et al., 2017</xref>; <xref ref-type="bibr" rid="B53">Xu et al., 2020</xref>).</p>
<p>Projecting hidden features from each model layer and the original features, clearly illustrated how low-level inputs are transformed into high-level representations that effectively distinguish tree species (section 4.5.2). In the shallow layers of model, input retains low-level features closely tied to their original forms, leading to mixed distributions of different tree species. As the model progresses to deeper layers, it captures features of increasing complexity most relevant to the categories, transforming hidden features to project different tree species into a more distinct feature space for precise separation (<xref ref-type="bibr" rid="B14">Goodfellow et al., 2016</xref>). We also found that the samples of deciduous species cluster closely after projection, as do the evergreen species, which aligns with our feature analysis, indicating smaller differences and reduced separability among similar tree species. In the final output layer, we observed significant inter-class separation among samples, indicating that deeper phenological features are captured by the model. Similarities lead to close proximity in the projected space, while differences create distinct clusters.</p>
</sec>
<sec id="S5.SS4">
<title>5.4 Tree species distribution and biodiversity</title>
<p>The tree species distribution map and forest biodiversity results revealed a certain correlation between tree species distribution and elevation (<xref ref-type="fig" rid="F6">Figure 6</xref>). <italic>Pinus tabuliformis</italic> dominates the forests of study area, primarily located in the low-elevation regions of eastern Taiyue Mountain and near the Huodong coal mining area, which also exhibited lower levels of biodiversity. According to information gathered from forestry departments, this trend may be attributed to large-scale plantings in the 1980s and reclamation plantings after mining activities, leading to a lack of species diversity. In contrast, Taiyue Mountain is predominantly covered by primary forests, which exhibited a richer species composition and higher levels of biodiversity. Many studies have found that primary forests have higher species diversity and biomass than planted and second-growth forests (<xref ref-type="bibr" rid="B4">Cavanaugh et al., 2014</xref>; <xref ref-type="bibr" rid="B43">Shirima et al., 2015</xref>). The second largest tree species in the study area, <italic>Quercus wutaishanica</italic>, is mainly in the southern part of Taiyue Mountain. <italic>Pinus bungeana</italic> is distributed along the western edge of Taiyue, forming a continuous strip, while <italic>Platycladus orientalis</italic> is located in the northwest, near the <italic>Pinus bungeana</italic>. <italic>Larix principis-rupprechtii</italic> and <italic>Quercus wutaishanica</italic> are concentrated in the high-altitude central region. The distribution and biodiversity data of tree species provided crucial information and effective support for forest ecosystem management and ecological research. For mining areas, where forest harvesting and replanting is a recurring process, it is crucial to implement more comprehensive and diverse planting strategies to protect and enhance both forest biodiversity and carbon sequestration capacity.</p>
</sec>
</sec>
<sec id="S6" sec-type="conclusion">
<title>6 Conclusion</title>
<p>In this study, we presented a deep learning-based method with SITS to achieve high-precision forest tree species and tree species biodiversity mapping. By pretraining the model on unlabeled forest pixels, the model performance was significantly enhanced, resulting in faster convergence and higher accuracy compared to the non-pretrained. Increasing the time series length improves classification accuracy, indicating that incorporating multi-temporal information across different phenological stages benefits tree species classification. While adding data beyond a full year can still improve model performance and confidence to some extent, the improvement is neither guaranteed nor limitless. Most vegetation indices, such as NDVIre, NDre and MSRre, are more sensitive than satellite bands in reflecting differences between tree species during key phenological stages, particularly between evergreen and deciduous trees. The Transformer-based model effectively captures and processes these key features, enabling accurate species classification. The methodologies designed for tree species classification and multidimensional interpretation, facilitate efficient species identification and enhance understanding of the integration of SITS and deep learning. This is valuable for related ecological research, and more studies are needed in the future to further explore the combination of SITS and deep learning to gain a clearer understanding of ecosystems.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="DS1">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="S8" sec-type="author-contributions">
<title>Author contributions</title>
<p>JT: Writing &#x2013; original draft, Formal Analysis, Software, Conceptualization, Writing &#x2013; review &#x0026; editing, Methodology. JL: Conceptualization, Supervision, Funding acquisition, Writing &#x2013; review &#x0026; editing. TM: Methodology, Validation, Writing &#x2013; review &#x0026; editing, Investigation. XY: Writing &#x2013; review &#x0026; editing, Data curation, Validation, Investigation. ZH: Writing &#x2013; review &#x0026; editing, Data curation, Validation.</p>
</sec>
<sec id="S9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the National Key Research and Development Program of China (2022YFE0127700).</p>
</sec>
<sec id="S10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="S11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="S12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="S13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/ffgc.2025.1599510/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/ffgc.2025.1599510/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="DS1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asner</surname> <given-names>G. P.</given-names></name> <name><surname>Jones</surname> <given-names>M. O.</given-names></name> <name><surname>Martin</surname> <given-names>R. E.</given-names></name> <name><surname>Knapp</surname> <given-names>D. E.</given-names></name> <name><surname>Hughes</surname> <given-names>R. F.</given-names></name></person-group> (<year>2008</year>). <article-title>Remote sensing of native and invasive species in Hawaiian forests.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>112</volume> <fpage>1912</fpage>&#x2013;<lpage>1926</lpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2007.02.043</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blickensd&#x00F6;rfer</surname> <given-names>L.</given-names></name> <name><surname>Oehmichen</surname> <given-names>K.</given-names></name> <name><surname>Pflugmacher</surname> <given-names>D.</given-names></name> <name><surname>Kleinschmit</surname> <given-names>B.</given-names></name> <name><surname>Hostert</surname> <given-names>P.</given-names></name></person-group> (<year>2024</year>). <article-title>National tree species mapping using Sentinel-1/2 time series and German National Forest Inventory data.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>304</volume>:<fpage>114069</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2024.114069</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Camps-Valls</surname> <given-names>G.</given-names></name> <name><surname>Bandos Marsheva</surname> <given-names>T. V.</given-names></name> <name><surname>Zhou</surname> <given-names>D.</given-names></name></person-group> (<year>2007</year>). <article-title>Semi-Supervised graph-based hyperspectral image classification.</article-title> <source><italic>IEEE Trans. Geosci. Remote Sens.</italic></source> <volume>45</volume> <fpage>3044</fpage>&#x2013;<lpage>3054</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2007.895416</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cavanaugh</surname> <given-names>K. C.</given-names></name> <name><surname>Gosnell</surname> <given-names>J. S.</given-names></name> <name><surname>Davis</surname> <given-names>S. L.</given-names></name> <name><surname>Ahumada</surname> <given-names>J.</given-names></name> <name><surname>Boundja</surname> <given-names>P.</given-names></name> <name><surname>Clark</surname> <given-names>D. B.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Carbon storage in tropical forests correlates with taxonomic diversity and functional dominance on a global scale.</article-title> <source><italic>Glob. Ecol. Biogeogr.</italic></source> <volume>23</volume> <fpage>563</fpage>&#x2013;<lpage>573</lpage>. <pub-id pub-id-type="doi">10.1111/geb.12143</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Qi</surname> <given-names>Z.</given-names></name> <name><surname>Shi</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Remote sensing image change detection with transformers.</article-title> <source><italic>IEEE Trans. Geosci. Remote Sens.</italic></source> <volume>60</volume> <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2021.3095166</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Bruin</surname> <given-names>S.</given-names></name> <name><surname>Brus</surname> <given-names>D. J.</given-names></name> <name><surname>Heuvelink</surname> <given-names>G. B. M.</given-names></name> <name><surname>Van Ebbenhorst Tengbergen</surname> <given-names>T.</given-names></name> <name><surname>Wadoux</surname> <given-names>A. M. J.-C.</given-names></name></person-group> (<year>2022</year>). <article-title>Dealing with clustered samples for assessing map accuracy by cross-validation.</article-title> <source><italic>Ecol. Inform.</italic></source> <volume>69</volume>:<fpage>101665</fpage>. <pub-id pub-id-type="doi">10.1016/j.ecoinf.2022.101665</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dou</surname> <given-names>P.</given-names></name> <name><surname>Shen</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Guan</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>Time series remote sensing image classification framework using combination of deep learning and multiple classifiers system.</article-title> <source><italic>Int. J. Appl. Earth Obs. Geoinform.</italic></source> <volume>103</volume>:<fpage>102477</fpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2021.102477</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fassnacht</surname> <given-names>F. E.</given-names></name> <name><surname>Latifi</surname> <given-names>H.</given-names></name> <name><surname>Stere&#x0144;czak</surname> <given-names>K.</given-names></name> <name><surname>Modzelewska</surname> <given-names>A.</given-names></name> <name><surname>Lefsky</surname> <given-names>M.</given-names></name> <name><surname>Waser</surname> <given-names>L. T.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Review of studies on tree species classification from remotely sensed data.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>186</volume> <fpage>64</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2016.08.013</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fassnacht</surname> <given-names>F. E.</given-names></name> <name><surname>Neumann</surname> <given-names>C.</given-names></name> <name><surname>Forster</surname> <given-names>M.</given-names></name> <name><surname>Buddenbaum</surname> <given-names>H.</given-names></name> <name><surname>Ghosh</surname> <given-names>A.</given-names></name> <name><surname>Clasen</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Comparison of feature reduction algorithms for classifying tree species with hyperspectral data on three central european test sites.</article-title> <source><italic>IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens.</italic></source> <volume>7</volume> <fpage>2547</fpage>&#x2013;<lpage>2561</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2014.2329390</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Felton</surname> <given-names>A.</given-names></name> <name><surname>Petersson</surname> <given-names>L.</given-names></name> <name><surname>Nilsson</surname> <given-names>O.</given-names></name> <name><surname>Witzell</surname> <given-names>J.</given-names></name> <name><surname>Cleary</surname> <given-names>M.</given-names></name> <name><surname>Felton</surname> <given-names>A. M.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>The tree species matters: Biodiversity and ecosystem service implications of replacing Scots pine production stands with Norway spruce.</article-title> <source><italic>Ambio</italic></source> <volume>49</volume> <fpage>1035</fpage>&#x2013;<lpage>1049</lpage>. <pub-id pub-id-type="doi">10.1007/s13280-019-01259-x</pub-id> <pub-id pub-id-type="pmid">31552644</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Foerster</surname> <given-names>S.</given-names></name> <name><surname>Kaden</surname> <given-names>K.</given-names></name> <name><surname>Foerster</surname> <given-names>M.</given-names></name> <name><surname>Itzerott</surname> <given-names>S.</given-names></name></person-group> (<year>2012</year>). <article-title>Crop type mapping using spectral&#x2013;temporal profiles and phenological information.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>89</volume> <fpage>30</fpage>&#x2013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2012.07.015</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frampton</surname> <given-names>W. J.</given-names></name> <name><surname>Dash</surname> <given-names>J.</given-names></name> <name><surname>Watmough</surname> <given-names>G.</given-names></name> <name><surname>Milton</surname> <given-names>E. J.</given-names></name></person-group> (<year>2013</year>). <article-title>Evaluating the capabilities of Sentinel-2 for quantitative estimation of biophysical variables in vegetation.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>82</volume> <fpage>83</fpage>&#x2013;<lpage>92</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2013.04.007</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>B.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Yao</surname> <given-names>H.</given-names></name> <name><surname>Liang</surname> <given-names>Y.</given-names></name> <name><surname>Deng</surname> <given-names>T.</given-names></name> <name><surname>He</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2022</year>). <article-title>Comparison of RFE-DL and stacking ensemble learning algorithms for classifying mangrove species on UAV multispectral images.</article-title> <source><italic>Int. J. Appl. Earth Obs. Geoinform.</italic></source> <volume>112</volume>:<fpage>102890</fpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2022.102890</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goodfellow</surname> <given-names>I.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Courville</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <source><italic>Deep learning, adaptive computation and machine learning.</italic></source> <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>The MIT Press</publisher-name>.</citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hemmerling</surname> <given-names>J.</given-names></name> <name><surname>Pflugmacher</surname> <given-names>D.</given-names></name> <name><surname>Hostert</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). <article-title>Mapping temperate forest tree species using dense Sentinel-2 time series.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>267</volume>:<fpage>112743</fpage>. <pub-id pub-id-type="doi">10.1016/6/j.rse.2021.112743</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hermosilla</surname> <given-names>T.</given-names></name> <name><surname>Bastyr</surname> <given-names>A.</given-names></name> <name><surname>Coops</surname> <given-names>N. C.</given-names></name> <name><surname>White</surname> <given-names>J. C.</given-names></name> <name><surname>Wulder</surname> <given-names>M. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Mapping the presence and distribution of tree species in Canada&#x2019;s forested ecosystems.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>282</volume>:<fpage>113276</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2022.113276</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>Q.</given-names></name> <name><surname>Sulla-Menashe</surname> <given-names>D.</given-names></name> <name><surname>Xu</surname> <given-names>B.</given-names></name> <name><surname>Yin</surname> <given-names>H.</given-names></name> <name><surname>Tang</surname> <given-names>H.</given-names></name> <name><surname>Yang</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>A phenology-based spectral and temporal feature selection method for crop mapping from satellite time series.</article-title> <source><italic>Int. J. Appl. Earth Obs. Geoinform.</italic></source> <volume>80</volume> <fpage>218</fpage>&#x2013;<lpage>229</lpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2019.04.014</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ienco</surname> <given-names>D.</given-names></name> <name><surname>Gaetano</surname> <given-names>R.</given-names></name> <name><surname>Dupaquier</surname> <given-names>C.</given-names></name> <name><surname>Maurel</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>Land cover classification via multitemporal spatial data by deep recurrent neural networks.</article-title> <source><italic>IEEE Geosci. Remote Sens. Lett.</italic></source> <volume>14</volume> <fpage>1685</fpage>&#x2013;<lpage>1689</lpage>. <pub-id pub-id-type="doi">10.1109/LGRS.2017.2728698</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Immitzer</surname> <given-names>M.</given-names></name> <name><surname>Neuwirth</surname> <given-names>M.</given-names></name> <name><surname>B&#x00F6;ck</surname> <given-names>S.</given-names></name> <name><surname>Brenner</surname> <given-names>H.</given-names></name> <name><surname>Vuolo</surname> <given-names>F.</given-names></name> <name><surname>Atzberger</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>Optimal input features for tree species classification in central Europe based on multi-temporal sentinel-2 data.</article-title> <source><italic>Remote Sens.</italic></source> <volume>11</volume>:<fpage>2599</fpage>. <pub-id pub-id-type="doi">10.3390/rs11222599</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Interdonato</surname> <given-names>R.</given-names></name> <name><surname>Ienco</surname> <given-names>D.</given-names></name> <name><surname>Gaetano</surname> <given-names>R.</given-names></name> <name><surname>Ose</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>DuPLO: A dual view point deep learning architecture for time series classification.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>149</volume> <fpage>91</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2019.01.011</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jing</surname> <given-names>L.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Self-Supervised visual feature learning with deep neural networks: A survey.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>43</volume> <fpage>4037</fpage>&#x2013;<lpage>4058</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.2992393</pub-id> <pub-id pub-id-type="pmid">32386141</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jolliffe</surname> <given-names>I. T.</given-names></name> <name><surname>Cadima</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Principal component analysis: A review and recent developments.</article-title> <source><italic>Philos. Trans. R. Soc. Math. Phys. Eng. Sci.</italic></source> <volume>374</volume>:<fpage>20150202</fpage>. <pub-id pub-id-type="doi">10.1098/rsta.2015.0202</pub-id> <pub-id pub-id-type="pmid">26953178</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kollert</surname> <given-names>A.</given-names></name> <name><surname>Bremer</surname> <given-names>M.</given-names></name> <name><surname>L&#x00F6;w</surname> <given-names>M.</given-names></name> <name><surname>Rutzinger</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Exploring the potential of land surface phenology and seasonal cloud free composites of one year of Sentinel-2 imagery for tree species mapping in a mountainous region.</article-title> <source><italic>Int. J. Appl. Earth Obs. Geoinform.</italic></source> <volume>94</volume>:<fpage>102208</fpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2020.102208</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>P.</given-names></name> <name><surname>Kumar</surname> <given-names>A.</given-names></name> <name><surname>Patil</surname> <given-names>M.</given-names></name> <name><surname>Hussain</surname> <given-names>S.</given-names></name> <name><surname>Singh</surname> <given-names>A. N.</given-names></name></person-group> (<year>2024</year>). <article-title>Factors influencing tree biomass and carbon stock in the Western Himalayas.</article-title> <source><italic>India. Front. For. Glob. Change</italic></source> <volume>6</volume>:<fpage>1328694</fpage>. <pub-id pub-id-type="doi">10.3389/ffgc.2023.1328694</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lechner</surname> <given-names>A. M.</given-names></name> <name><surname>Foody</surname> <given-names>G. M.</given-names></name> <name><surname>Boyd</surname> <given-names>D. S.</given-names></name></person-group> (<year>2020</year>). <article-title>Applications in remote sensing to forest ecology and management.</article-title> <source><italic>One Earth</italic></source> <volume>2</volume> <fpage>405</fpage>&#x2013;<lpage>412</lpage>. <pub-id pub-id-type="doi">10.1016/j.oneear.2020.05.001</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Miao</surname> <given-names>Y.</given-names></name> <name><surname>Feng</surname> <given-names>G.</given-names></name> <name><surname>Yuan</surname> <given-names>F.</given-names></name> <name><surname>Yue</surname> <given-names>S.</given-names></name> <name><surname>Gao</surname> <given-names>X.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Improving estimation of summer maize nitrogen status with red edge-based spectral vegetation indices.</article-title> <source><italic>Field Crops Res.</italic></source> <volume>157</volume> <fpage>111</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/j.fcr.2013.12.018</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Zhao</surname> <given-names>W.</given-names></name> <name><surname>Peng</surname> <given-names>R.</given-names></name> <name><surname>Ye</surname> <given-names>T.</given-names></name></person-group> (<year>2022</year>). <article-title>Multi-branch self-learning Vision Transformer (MSViT) for crop type mapping with Optical-SAR time-series.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>203</volume>:<fpage>107497</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2022.107497</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lindner</surname> <given-names>M.</given-names></name> <name><surname>Maroschek</surname> <given-names>M.</given-names></name> <name><surname>Netherer</surname> <given-names>S.</given-names></name> <name><surname>Kremer</surname> <given-names>A.</given-names></name> <name><surname>Barbati</surname> <given-names>A.</given-names></name> <name><surname>Garcia-Gonzalo</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2010</year>). <article-title>Climate change impacts, adaptive capacity, and vulnerability of European forest ecosystems.</article-title> <source><italic>For. Ecol. Manag.</italic></source> <volume>259</volume> <fpage>698</fpage>&#x2013;<lpage>709</lpage>. <pub-id pub-id-type="doi">10.1016/j.foreco.2009.09.023</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lipton</surname> <given-names>Z. C.</given-names></name></person-group> (<year>2018</year>). <article-title>The mythos of model interpretability.</article-title> <source><italic>Commun. ACM</italic></source> <volume>61</volume> <fpage>36</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1145/3233231</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Sun</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name><etal/></person-group> (<year>2023</year>). <article-title>Comparisons between temporal statistical metrics, time series stacks and phenological features derived from NASA Harmonized Landsat Sentinel-2 data for crop type mapping.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>211</volume>:<fpage>108015</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2023.108015</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>L&#x00F6;w</surname> <given-names>F.</given-names></name> <name><surname>Michel</surname> <given-names>U.</given-names></name> <name><surname>Dech</surname> <given-names>S.</given-names></name> <name><surname>Conrad</surname> <given-names>C.</given-names></name></person-group> (<year>2013</year>). <article-title>Impact of feature selection on the accuracy and spatial uncertainty of per-field crop classification using support vector machines.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>85</volume> <fpage>102</fpage>&#x2013;<lpage>119</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2013.08.007</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Melnyk</surname> <given-names>O.</given-names></name> <name><surname>Manko</surname> <given-names>P.</given-names></name> <name><surname>Brunn</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Remote sensing methods for estimating tree species of forests in the Volyn region, Ukraine.</article-title> <source><italic>Front. For. Glob. Change</italic></source> <volume>6</volume>:<fpage>1041882</fpage>. <pub-id pub-id-type="doi">10.3389/ffgc.2023.1041882</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Miller</surname> <given-names>L.</given-names></name> <name><surname>Pelletier</surname> <given-names>C.</given-names></name> <name><surname>Webb</surname> <given-names>G. I.</given-names></name></person-group> (<year>2024</year>). <article-title>Deep learning for satellite image time-series analysis: A review.</article-title> <source><italic>IEEE Geosci. Remote Sens. Mag.</italic></source> <volume>12</volume> <fpage>81</fpage>&#x2013;<lpage>124</lpage>. <pub-id pub-id-type="doi">10.1109/MGRS.2024.3393010</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Misra</surname> <given-names>I.</given-names></name> <name><surname>Van Der Maaten</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>Self-Supervised learning of pretext-invariant representations</article-title>,&#x201D; in <source><italic>Proceedings of the 2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ngo</surname> <given-names>Y.-N.</given-names></name> <name><surname>Ho Tong, Minh</surname> <given-names>D.</given-names></name> <name><surname>Baghdadi</surname> <given-names>N.</given-names></name> <name><surname>Fayad</surname> <given-names>I.</given-names></name></person-group> (<year>2023</year>). <article-title>Tropical forest top height by GEDI: From sparse coverage to continuous data.</article-title> <source><italic>Remote Sens.</italic></source> <volume>15</volume>:<fpage>975</fpage>. <pub-id pub-id-type="doi">10.3390/rs15040975</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pedregosa</surname> <given-names>F.</given-names></name> <name><surname>Varoquaux</surname> <given-names>G.</given-names></name> <name><surname>Gramfort</surname> <given-names>A.</given-names></name> <name><surname>Michel</surname> <given-names>V.</given-names></name> <name><surname>Thirion</surname> <given-names>B.</given-names></name> <name><surname>Grisel</surname> <given-names>O.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>Scikit-learn: Machine learning in python.</article-title> <source><italic>J. Mach. Learn. Res.</italic></source> <volume>12</volume> <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>. <pub-id pub-id-type="doi">10.5555/1953048.2078195</pub-id> <pub-id pub-id-type="pmid">34820480</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pelletier</surname> <given-names>C.</given-names></name> <name><surname>Webb</surname> <given-names>G.</given-names></name> <name><surname>Petitjean</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>Temporal convolutional neural network for the classification of satellite image time series.</article-title> <source><italic>Remote Sens.</italic></source> <volume>11</volume>:<fpage>523</fpage>. <pub-id pub-id-type="doi">10.3390/rs11050523</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Modelling of the biodiversity of tropical forests in China based on unmanned aerial vehicle multispectral and light detection and ranging data.</article-title> <source><italic>Int. J. Remote Sens.</italic></source> <volume>42</volume> <fpage>8858</fpage>&#x2013;<lpage>8877</lpage>. <pub-id pub-id-type="doi">10.1080/01431161.2021.1954714</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ru&#x00DF;wurm</surname> <given-names>M.</given-names></name> <name><surname>Korner</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Temporal vegetation modelling using long short-term memory networks for crop identification from medium-resolution multi-spectral satellite images</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</italic></source>, (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ru&#x00DF;wurm</surname> <given-names>M.</given-names></name> <name><surname>K&#x00F6;rner</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Self-attention for raw optical satellite time series classification.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>169</volume> <fpage>421</fpage>&#x2013;<lpage>435</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2020.06.006</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Samek</surname> <given-names>W.</given-names></name> <name><surname>Binder</surname> <given-names>A.</given-names></name> <name><surname>Montavon</surname> <given-names>G.</given-names></name> <name><surname>Lapuschkin</surname> <given-names>S.</given-names></name> <name><surname>Muller</surname> <given-names>K.-R.</given-names></name></person-group> (<year>2017</year>). <article-title>Evaluating the visualization of what a deep neural network has learned.</article-title> <source><italic>IEEE Trans. Neural Netw. Learn. Syst.</italic></source> <volume>28</volume> <fpage>2660</fpage>&#x2013;<lpage>2673</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2016.2599820</pub-id> <pub-id pub-id-type="pmid">27576267</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schulz</surname> <given-names>C.</given-names></name> <name><surname>F&#x00F6;rster</surname> <given-names>M.</given-names></name> <name><surname>Vulova</surname> <given-names>S. V.</given-names></name> <name><surname>Rocha</surname> <given-names>A. D.</given-names></name> <name><surname>Kleinschmit</surname> <given-names>B.</given-names></name></person-group> (<year>2024</year>). <article-title>Spectral-temporal traits in Sentinel-1 C-band SAR and Sentinel-2 multispectral remote sensing time series for 61 tree species in Central Europe.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>307</volume>:<fpage>114162</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2024.114162</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shirima</surname> <given-names>D. D.</given-names></name> <name><surname>Totland</surname> <given-names>&#x00D8;</given-names></name> <name><surname>Munishi</surname> <given-names>P. K. T.</given-names></name> <name><surname>Moe</surname> <given-names>S. R.</given-names></name></person-group> (<year>2015</year>). <article-title>Relationships between tree species richness, evenness and aboveground carbon storage in montane forests and miombo woodlands of Tanzania.</article-title> <source><italic>Basic Appl. Ecol.</italic></source> <volume>16</volume> <fpage>239</fpage>&#x2013;<lpage>249</lpage>. <pub-id pub-id-type="doi">10.1016/j.baae.2014.11.008</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Somers</surname> <given-names>B.</given-names></name> <name><surname>Asner</surname> <given-names>G. P.</given-names></name></person-group> (<year>2014</year>). <article-title>Tree species mapping in tropical forests using multi-temporal imaging spectroscopy: Wavelength adaptive spectral mixture analysis.</article-title> <source><italic>Int. J. Appl. Earth Obs. Geoinform.</italic></source> <volume>31</volume> <fpage>57</fpage>&#x2013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2014.02.006</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Somers</surname> <given-names>B.</given-names></name> <name><surname>Delalieux</surname> <given-names>S.</given-names></name> <name><surname>Verstraeten</surname> <given-names>W. W.</given-names></name> <name><surname>Van Aardt</surname> <given-names>J. A. N.</given-names></name> <name><surname>Albrigo</surname> <given-names>G. L.</given-names></name> <name><surname>Coppin</surname> <given-names>P.</given-names></name></person-group> (<year>2010</year>). <article-title>An automated waveband selection technique for optimized hyperspectral mixture analysis.</article-title> <source><italic>Int. J. Remote Sens.</italic></source> <volume>31</volume> <fpage>5549</fpage>&#x2013;<lpage>5568</lpage>. <pub-id pub-id-type="doi">10.1080/01431160903311305</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>K.</given-names></name> <name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Du</surname> <given-names>P.</given-names></name></person-group> (<year>2015</year>). <article-title>A novel semi-supervised hyperspectral image classification approach based on spatial neighborhood information and classifier combination.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>105</volume> <fpage>19</fpage>&#x2013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2015.03.006</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2008</year>). <article-title>Visualizing data using t-SNE.</article-title> <source><italic>J. Mach. Learn. Res.</italic></source> <volume>9</volume> <fpage>2579</fpage>&#x2013;<lpage>2605</lpage>.</citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Parmar</surname> <given-names>N.</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J.</given-names></name> <name><surname>Jones</surname> <given-names>L.</given-names></name> <name><surname>Gomez</surname> <given-names>A. N.</given-names></name><etal/></person-group> (<year>2017</year>). &#x201C;<article-title>Attention is all you need</article-title>,&#x201D; in <source><italic>Proceedings of the 31st International Conference on Neural Information Processing Systems (NeurIPS 2017)</italic></source>, (<publisher-loc>California, CA</publisher-loc>: <publisher-name>Long Beach</publisher-name>).</citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>R.</given-names></name> <name><surname>Gamon</surname> <given-names>J. A.</given-names></name></person-group> (<year>2019</year>). <article-title>Remote sensing of terrestrial plant biodiversity.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>231</volume>:<fpage>111218</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2019.111218</pub-id></citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>White</surname> <given-names>J. C.</given-names></name> <name><surname>Coops</surname> <given-names>N. C.</given-names></name> <name><surname>Wulder</surname> <given-names>M. A.</given-names></name> <name><surname>Vastaranta</surname> <given-names>M.</given-names></name> <name><surname>Hilker</surname> <given-names>T.</given-names></name> <name><surname>Tompalski</surname> <given-names>P.</given-names></name></person-group> (<year>2016</year>). <article-title>Remote sensing technologies for enhancing forest inventories: A review.</article-title> <source><italic>Can. J. Remote Sens.</italic></source> <volume>42</volume> <fpage>619</fpage>&#x2013;<lpage>641</lpage>. <pub-id pub-id-type="doi">10.1080/07038992.2016.1207484</pub-id></citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>J.</given-names></name> <name><surname>Chevallier</surname> <given-names>F.</given-names></name> <name><surname>Gomez</surname> <given-names>C.</given-names></name> <name><surname>Guanter</surname> <given-names>L.</given-names></name> <name><surname>Hicke</surname> <given-names>J. A.</given-names></name> <name><surname>Huete</surname> <given-names>A. R.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Remote sensing of the terrestrial carbon cycle: A review of advances over 50 years.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>233</volume>:<fpage>111383</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2019.111383</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Xiong</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Ting</surname> <given-names>K. C.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Towards interpreting multi-temporal deep learning models in crop mapping.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>264</volume>:<fpage>112599</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2021.112599</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Zhong</surname> <given-names>R.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>DeepCropMapping: A multi-temporal deep learning approach with improved spatial generalizability for dynamic corn and soybean mapping.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>247</volume>:<fpage>111946</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2020.111946</pub-id></citation></ref>
<ref id="B54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>B.</given-names></name> <name><surname>Wu</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>T.</given-names></name></person-group> (<year>2024</year>). <article-title>Mapping forest tree species using sentinel-2 time series by taking into account tree age.</article-title> <source><italic>Forests</italic></source> <volume>15</volume>:<fpage>474</fpage>. <pub-id pub-id-type="doi">10.3390/f15030474</pub-id></citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>You</surname> <given-names>N.</given-names></name> <name><surname>Dong</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Examining earliest identifiable timing of crops using all available sentinel 1/2 imagery and google earth engine.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>161</volume> <fpage>109</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2020.01.001</pub-id></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yuan</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>Self-Supervised pretraining of transformers for satellite image time series classification.</article-title> <source><italic>IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens.</italic></source> <volume>14</volume> <fpage>474</fpage>&#x2013;<lpage>487</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2020.3036602</pub-id></citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>N.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>A new strategy for improving the accuracy of forest aboveground biomass estimates in an alpine region based on multi-source remote sensing.</article-title> <source><italic>GIScience Remote Sens.</italic></source> <volume>60</volume>:<fpage>2163574</fpage>. <pub-id pub-id-type="doi">10.1080/15481603.2022.2163574</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>W.</given-names></name> <name><surname>Qu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name></person-group> (<year>2022</year>). <article-title>Spatial-aware SAR-optical time-series deep integration for crop phenology tracking.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>276</volume>:<fpage>113046</fpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2022.113046</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhong</surname> <given-names>L.</given-names></name> <name><surname>Hu</surname> <given-names>L.</given-names></name> <name><surname>Zhou</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>Deep learning based multi-temporal crop classification.</article-title> <source><italic>Remote Sens. Environ.</italic></source> <volume>221</volume> <fpage>430</fpage>&#x2013;<lpage>443</lpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2018.11.032</pub-id></citation></ref>
<ref id="B60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z.-H.</given-names></name></person-group> (<year>2018</year>). <article-title>A brief introduction to weakly supervised learning.</article-title> <source><italic>Natl. Sci. Rev.</italic></source> <volume>5</volume> <fpage>44</fpage>&#x2013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1093/nsr/nwx106</pub-id></citation></ref>
</ref-list>
</back>
</article>