<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd"> 
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2023.1130659</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Cropformer: A new generalized deep learning classification approach for multi-scenario crop classification</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Hengbin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2148591"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chang</surname>
<given-names>Wanqiu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yao</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2147321"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yao</surname>
<given-names>Zhiying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2152465"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhao</surname>
<given-names>Yuanyuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1415033"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Shaoming</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Zhe</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1567891"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Xiaodong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Land Science and Technology, China Agricultural University</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Key Laboratory of Remote Sensing for Agri-Hazards, Ministry of Agriculture and Rural Affairs</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Muhammad Fazal Ijaz, Sejong University, Republic of Korea</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Parvathaneni Naga Srinivasu, Prasad V. Potluri Siddhartha Institute of Technology, India; Jana Shafi, Prince Sattam Bin Abdulaziz University, Saudi Arabia; Farman Ali, Sejong University, Republic of Korea</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yuanyuan Zhao, <email xlink:href="mailto:zhaoyuanyuan@cau.edu.cn">zhaoyuanyuan@cau.edu.cn</email>
</p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Technical Advances in Plant Science, a section of the journal Frontiers in Plant Science</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>03</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1130659</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>02</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Wang, Chang, Yao, Yao, Zhao, Li, Liu and Zhang</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Wang, Chang, Yao, Yao, Zhao, Li, Liu and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Accurate and efficient crop classification using remotely sensed data can provide fundamental and important information for crop yield estimation. Existing crop classification approaches are usually designed to be strong in some specific scenarios but not for multi-scenario crop classification. In this study, we proposed a new deep learning approach for multi-scenario crop classification, named Cropformer. Cropformer can extract global features and local features, to solve the problem that current crop classification methods extract a single feature. Specifically, Cropformer is a two-step classification approach, where the first step is self-supervised pre-training to accumulate knowledge of crop growth, and the second step is a fine-tuned supervised classification based on the weights from the first step. The unlabeled time series and the labeled time series are used as input for the first and second steps respectively. Multi-scenario crop classification experiments including full-season crop classification, in-season crop classification, few-sample crop classification, and transfer of classification models were conducted in five study areas with complex crop types and compared with several existing competitive approaches. Experimental results showed that Cropformer can not only obtain a very significant accuracy advantage in crop classification, but also can obtain higher accuracy with fewer samples. Compared to other approaches, the classification performance of Cropformer during model transfer and the efficiency of the classification were outstanding. The results showed that Cropformer could build up <italic>a priori</italic> knowledge using unlabeled data and learn generalized features using labeled data, making it applicable to crop classification in multiple scenarios.</p>
</abstract>
<kwd-group>
<kwd>multi-scenario crop classification</kwd>
<kwd>time series</kwd>
<kwd>deep learning</kwd>
<kwd>pre-training</kwd>
<kwd>Cropformer</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="5"/>
<equation-count count="7"/>
<ref-count count="68"/>
<page-count count="20"/>
<word-count count="10342"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>With a large amount of remotely sensed data available free, easily, and quickly, remote sensing plays an increasingly important role in vegetation and land cover mapping. The timely and accurate vegetation/land cover information generated by remote sensing images can provide important data for resource management, ecological monitoring, agricultural production, and other fields. For large-scale crop classification, continuous and full-coverage satellite images provided by remote sensing are particularly important (<xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B65">Zhang et&#xa0;al., 2020</xref>), and how to extract useful information for crop classification from satellite images has been explored. In addition, a large number of machine learning algorithms have been introduced to obtain large-scale crop distribution more accurately (<xref ref-type="bibr" rid="B1">Abdullah et&#xa0;al., 2019</xref>).</p>
<p>Full-season crop classification is the most common scenario in which current crop classification approaches are applied, but it is generally available at the end of the growing season. In-season crop classification allows the distribution of crops to be obtained as early as the growing season, which is of interest for agricultural production guidance, but few publicly available data are available(<xref ref-type="bibr" rid="B59">Xu et&#xa0;al., 2020</xref>). Current research has focused on classification approaches supported by large numbers of samples (<xref ref-type="bibr" rid="B60">Yi et&#xa0;al., 2020</xref>). Crop classification in a few-sample context is of interest in the crop classification scenario, where obtaining highly accurate classification results with very few samples can reduce costs. Crop classification in regions with no samples is difficult, and it is feasible to train a well-trained model in a sample-rich region and transfer it to a region with no samples (<xref ref-type="bibr" rid="B14">Hao et&#xa0;al., 2020</xref>). The existing crop classification methods all focus on one application scenario or two application scenarios, and there is no discussion of crop classification for multiple application scenarios, nor is there a general crop classification model that can be used. On this basis, developing a classification approach that can be applied to the above classification scenario is greatly needed.</p>
<p>Traditional crop classifiers include Decision Trees (DT), Random Forests (RF), and Support Vector Machines (SVM) (<xref ref-type="bibr" rid="B31">Low et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B23">Khatami et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B46">Shi and Yang, 2016</xref>). The inputs to these classifiers are usually manually designed features including spectral values, vegetation indices, etc., and multi-temporal satellite observations are used instead of mono-temporal imagery (<xref ref-type="bibr" rid="B12">Feng et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B11">Eudes Gbodjo et&#xa0;al., 2020</xref>). Although multi-temporal inputs are effective in improving classification performance, these classification models often ignore temporal dependence in the time series. Traditional methods require manual design of inputs, which need to be designed differently for different application scenarios. However, the manually designed features have some limitations and are very dependent on <italic>a priori</italic> knowledge and expertise, and the complex changes in realistic conditions affect the manually designed features more, which makes the classification models less robust and less generalizable.</p>
<p>In contrast to classical machine learning, deep learning no longer requires manually designed features, but can learn complex semantic features from high-dimensional data. Currently, deep learning has been widely used in agriculture due to its effectiveness, including crop classification (<xref ref-type="bibr" rid="B24">Kussul et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B34">Minh et&#xa0;al., 2018</xref>), pest and disease detection (<xref ref-type="bibr" rid="B2">Akbar et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B47">Shoaib et&#xa0;al., 2022a</xref>; <xref ref-type="bibr" rid="B48">Shoaib et&#xa0;al., 2022b</xref>), yield estimation(<xref ref-type="bibr" rid="B37">Nevavuori et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B22">Khaki et&#xa0;al., 2020</xref>), etc. The complex network structure of deep learning requires a large amount of labeled data for support, which creates difficulties for the agricultural fields where deep learning is applied. When the sample size is insufficient, it makes the model very easy to over fit and thus the application is much less meaningful. For agriculture, especially crop classification, the acquisition of labeled samples is not as easy as in computer vision, and each labeled sample acquisition is resource-intensive (<xref ref-type="bibr" rid="B59">Xu et&#xa0;al., 2020</xref>). Therefore, it is necessary to address the problem of deep learning in crop classification that requires the use of a large number of samples. Also, developing a generalized deep learning model that can use only a small number of samples and can be applied to other crop classification scenarios is a scientific challenge.</p>
<p>Multi-temporal observations add a more intensive focus on the phonological cycle of crop growth, but also bring the problem of not keeping the same interval between observations in the study area, which makes the acquired time series irregular. Irregular time series are not directly usable for RF, SVM, and need to be normalized to obtain regular time series. Standardization methods include the rejection of invalid spectral values (<xref ref-type="bibr" rid="B1">Abdullah et&#xa0;al., 2019</xref>), missing spectral value supplementation (<xref ref-type="bibr" rid="B19">Ienco et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B24">Kussul et&#xa0;al., 2017</xref>), and spectral value resampling (<xref ref-type="bibr" rid="B29">Liu et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B58">Wang et&#xa0;al., 2020</xref>), but this standardization process changes the original sequence information as well as increase the computational effort. Direct use of irregular time series can avoid the standardization process, but it also causes a decrease in classification accuracy. Dynamic Time Warping (DTW) has been used for the analysis of irregular time series, but its traversal algorithm can significantly increase the computational effort (<xref ref-type="bibr" rid="B40">Petitjean et&#xa0;al., 2012</xref>). The introduction of a Gaussian process to solve irregular time sampling and missing data is robust, but does not compare favorably with other methods in terms of classification performance (<xref ref-type="bibr" rid="B7">Constantin et&#xa0;al., 2021</xref>). The existing approaches using irregular time series usually involve two aspects. On the one hand, it starts from the time series itself, but this approach changes the original growth pattern of the crop, making it difficult for the model to learn the original growth pattern of the crop. On the other hand, it starts from the method of processing time series, which can directly use irregular time series but cannot form a complete system and will significantly increase the workload. Therefore, current methods either do not achieve satisfactory accuracy or do not allow the use of end-to-end classification methods.</p>
<p>The unlabeled data contain rich crop growth information, and unknown crop growth information can be used as <italic>a priori</italic> knowledge. Pre-training as an effective training method has been applied to land use classification to accelerate the convergence of the training process (<xref ref-type="bibr" rid="B67">Zhao et&#xa0;al., 2017</xref>). The self-supervised pre-training approach can improve the utilization of labeled samples in land cover classification (<xref ref-type="bibr" rid="B53">Tarasiou and Zafeiriou, 2022</xref>). Unlabeled data as pre-training data can effectively improve crop classification accuracy and reduce the use of labeled samples (<xref ref-type="bibr" rid="B61">Yuan and Lin, 2021</xref>; <xref ref-type="bibr" rid="B62">Yuan et&#xa0;al., 2022</xref>). However, pre-training is still less used in scenarios such as in-season crop classification and model transfer. At present, there is also no general pre-trained classification model that can be applied to multi-scenario crop classification.</p>
<p>This study aims to build a deep learning classification model that can be generalized in multi-scenario crop classification. The potential of a pre-trained classification model based on Transformer and Convolution structures for application in multi-scenario crop classification is evaluated. In this paper, we proposed a new deep learning approach, Cropformer, for multi-scenario crop classification. Full-season crop classification, in-season crop classification, few-sample crop classification, and model transfer experiments were set up in five study areas rich in crop types. A variety of best existing classifications were compared with different indicators. Our novel contributions are threefold:</p>
<list list-type="simple">
<list-item>
<p>1. A deep learning structure that fuses Convolution and Transformer is designed. The Transformer captures features throughout the reproduction period of the crop, while the convolution effectively utilizes information from key growth nodes of the crop. Combining the two features can improve the generalization ability of the model, which can be applied to multi-scenario crop classification.</p>
</list-item>
<list-item>
<p>2. The introduction of time-dimensional features on the input side of the model increases the diversity of the input. Position encoding has been added to solve the problem of unusable irregular time series due to missing values in remote sensing imagery.</p>
</list-item>
<list-item>
<p>3. Using a two-step classification framework and pre-training with unlabeled data increases the accumulation of crop growth knowledge in the model and improves the adaptability of the model in multi-scene crop classification.</p>
</list-item>
</list>
<p>The remainder of this article is organized as follows. Section II summarizes related work on crop classification. Section III provides a description of the remote sensing images and samples used in this paper. Section IV explains the motivation of the proposed method and describes the proposed network architecture. Section V reports the experimental results. Section VI discusses the article and presents future work. Finally, Section VII concludes this article.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<p>An effective and general classification method is a prerequisite for achieving high accuracy crop classification. The more comprehensive the features extracted by the classification model, the more significant the advantages of the classification results obtained. According to the differences in feature selection strategies, existing crop classification methods can be classified into the following three categories.</p>
<p>
<italic>Supervised traditional classification methods</italic> include machine learning classification models, such as SVM, RF, and Multilayer Perceptual (MLP). These models are sensitive to the spectral information of crops and use vegetation indices and spectral values as the main feature inputs. However, the sequence relationships hidden in the time series are not exploited, so more temporal features are incorporated in the model inputs including the statistical value of spectral value and vegetation indices and statistical features of vegetation indices curve (<xref ref-type="bibr" rid="B39">Pelletier et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B66">Zhang et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B63">Zeng et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B30">Liu et&#xa0;al., 2021</xref>). Comparing multiple crop vegetation index curves and obtaining key dates and key observations from the curves to distinguish crops can be effective in improving classification performance (<xref ref-type="bibr" rid="B49">Simonneaux et&#xa0;al., 2008</xref>; <xref ref-type="bibr" rid="B26">Lebourgeois et&#xa0;al., 2017</xref>). Many studies have used various functions to fit crop growth characteristics, including wavelet transform and double logistic function, and used the main parameters and significant stages of the functions as features for classification (<xref ref-type="bibr" rid="B44">Sakamoto et&#xa0;al., 2006</xref>; <xref ref-type="bibr" rid="B50">Soudani et&#xa0;al., 2008</xref>). Supervised traditional classification methods do not require a complex feature extraction process and the time cost is substantial. However, supervised traditional classification methods are influenced by their inputs as well as feature extraction strategies, and their application scenarios are single and cannot be adapted to multi- scenario crop classification.</p>
<p>
<italic>Supervised deep learning classification methods</italic> include two outstanding algorithms Recurrent Neural Networks (RNN) and Convolutional Neural Networks (CNN) that can efficiently process sequential data (<xref ref-type="bibr" rid="B19">Ienco et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B34">Minh et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B68">Zhong et&#xa0;al., 2019</xref>). RNN has the unique advantage of processing sequential data, which is sensitive to temporal order (Mou and Zhu, 2018; <xref ref-type="bibr" rid="B45">Sharma et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B38">Papadomanolaki et&#xa0;al., 2019</xref>). Long Short-Term Memory (LSTM) model handles longer time series than RNN and has proven its effectiveness in capturing features in several classification models (<xref ref-type="bibr" rid="B43">Ru&#xdf;wurm and Korner, 2017</xref>; <xref ref-type="bibr" rid="B68">Zhong et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B41">Rajendran et&#xa0;al., 2020</xref>). The advantages of LSTM for inter-annual samples and the spatial transfer capability were also confirmed (<xref ref-type="bibr" rid="B59">Xu et&#xa0;al., 2020</xref>). LSTM has a drawback in parallel computation and cannot compute multiple layers at the same time, which can significantly increase the time consumption and is not practical for large-scale crop mapping. CNN has sparse swapping and parameter sharing, which may be able to reduce the time of network training (<xref ref-type="bibr" rid="B52">Tai et&#xa0;al., 2017</xref>). Different forms of input design have an important impact on classification performance for CNN (<xref ref-type="bibr" rid="B32">Marcos et&#xa0;al., 2018</xref>). From one-dimensional sequences, to two-dimensional images, to three-dimensional video streams have been used as input to CNN for hyperspectral or multispectral data classification (<xref ref-type="bibr" rid="B4">Chen et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B24">Kussul et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B18">Huang et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B21">Ji et&#xa0;al., 2018</xref>), but the conversion of multi-temporal observations to image or video streams increases the classification cost. In addition, a novel network structure combining RNN and CNN has been proposed to extract temporal features by learning temporal correlations (<xref ref-type="bibr" rid="B35">Mou et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B33">Martinez et&#xa0;al., 2021</xref>), but the features extracted by RNN and CNN are local features, which prevents the model from learning features from a global perspective. Therefore, the current supervised deep learning classification methods still have the drawback of extracting a single feature. Feature extraction is performed only from one side of crop growth, without combining key nodes of crop growth stages and the whole reproductive period, and it is impossible to obtain diverse features that can characterize crop growth patterns.</p>
<p>
<italic>Self-Supervised deep learning classification methods</italic> have gained much attention because of its excellent learning ability on unlabeled data. The Transformer, consisting of multiple self-attention structures, is currently the most commonly used SSL structure (<xref ref-type="bibr" rid="B55">Vaswani et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B9">Devlin et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B10">Dosovitskiy et&#xa0;al., 2020</xref>). Xu et&#xa0;al. (<xref ref-type="bibr" rid="B59">Xu et&#xa0;al., 2020</xref>) verified that Transformer has advantages for processing time series, using parallel operations to overcome the problem of time-consuming processing of long time series, but experiments have only been conducted in in-season and full-season crop mapping. The Transformer is good at capturing global information but not sensitive to local information, so fusing the Transformer and CNN into a new network structure becomes a popular way (<xref ref-type="bibr" rid="B13">Gulati et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B64">Zhang et&#xa0;al., 2022</xref>). <xref ref-type="bibr" rid="B27">Li et&#xa0;al. (2020)</xref> developed a hybrid Convolution and Transformer network structure for multi-source remote sensing image classification and verified the feasibility of fusing the two, but it only discussed the effectiveness of the hybrid structure in full-season crop classification. However, the potential of the new network structure combining Transformer and CNN in the field of crop classification in other scenarios has not been fully tested. A general model with the ability to capture both global and local information is necessary for multi-scenario crop classification.</p>
<p>In conclusion, current crop classification methods still suffer from insufficient feature extraction ability, single application scenario, and lack of a general and effective classification method.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Materials</title>
<sec id="s3_1">
<label>3.1</label>
<title>Study area</title>
<p>In this paper, three of the five study areas are located in Northwest China and two in Northeast China, as shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. The first study area is the Hexi Corridor, which is an important agricultural production area with abundant light resources and abundant snow and ice meltwater. The second study area is the Ili River Valley, which has a mild climate, a temperate continental climate, abundant sunshine and precipitation, and significant advantages for agricultural development. The third study area is the Tianshan Corridor, which is in the mid-temperate arid climate zone and is an oasis irrigated agricultural area. The fourth and fifth study areas are Western and Eastern Heilongjiang, which have a temperate continental climate and whose black land advantage makes them an important food supply base for China. These five regions have favorable agricultural production conditions and diverse crop types, which are ideal for verifying the validity as well as the robustness of the classification model.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The geographical locations of the five study areas in China. The data used for pre-training are from the white boxed area in the Ili River Valley, Western Heilongjiang and Eastern Heilongjiang. The lower part shows a remote sensing image of GaoFen 1.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g001.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Satellite imagery</title>
<p>The remote sensing images we used are acquired by the GaoFen-1 satellite, which contains four bands: red, green, blue, and near-infrared, with a spatial resolution of 16&#xa0;m and a temporal resolution of 4 days. We only used images with cloud coverage of less than 10%, which causes inconsistencies in the length of the time series. The number of valid images for each study area is shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, where Pre_His and Pre_Cur represent the imagery covering the pre-training sample areas acquired in the historical year and current year, respectively. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> shows that the number of remotely sensed images in each region and each month is different, and directly using the time series obtained from the images as input is a challenge for most models, which require pre-processing of the extracted time series. Cropformer, on the other hand, does not require any processing and can directly use the time series extracted from the images, which greatly improves the efficiency of the classification. The way of dealing with the irregular time series inputs by Cropformer is described in Section 4.1.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Number of valid images in the five study areas.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="left">Study Area</th>
<th valign="top" align="center">Mar.</th>
<th valign="top" align="center">Apr.</th>
<th valign="top" align="center">May.</th>
<th valign="top" align="center">Jun.</th>
<th valign="top" align="center">Jul.</th>
<th valign="top" align="center">Aug.</th>
<th valign="top" align="center">Sep.</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="3" align="left"/>
<td valign="top" align="left">I</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">45</td>
</tr>
<tr>
<td valign="top" align="left">II</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">27</td>
</tr>
<tr>
<td valign="top" align="left">III</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">54</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">IV</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">24</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">V</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">12</td>
</tr>
<tr>
<td valign="top" align="left">Irregular</td>
<td valign="top" align="left">Pre_His_IV</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">16</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">Pre_His_V</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">6</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">Pre_His_II</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">23</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">Pre_Cur_II</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">12</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">Pre_Cur_IV</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">19</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">Pre_Cur_V</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">11</td>
</tr>
<tr>
<td valign="top" align="left">Regular</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">22</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Sample dataset</title>
<p>Each of the three study areas of Northwest China has about 20 crop types in each region, and the main crops are maize, cotton, grape, and wheat. The two study areas in Northeastern China have a stable cropping structure, with the main crops being maize, rice, and soybean. We used unlabeled data from two 10&#xa0;km &#xd7; 10&#xa0;km areas as the pre-training dataset, with a total of 781,250 sample units. In addition, field samples from three study areas collected in the current year were used for training and testing, where the ratio of the training dataset, validation dataset, and testing dataset was 6:2:2. The distribution of sample categories in the three study areas of Northwest China was very unbalanced, which also challenged our classification model. The numbers of sample units in each crop type are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Number of Samples in the Three Study Areas.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Study area</th>
<th valign="middle" align="center">Crop type</th>
<th valign="middle" align="center">Number of samples</th>
<th valign="middle" align="center">Study area</th>
<th valign="middle" align="center">Crop type</th>
<th valign="middle" align="center">Number of samples</th>
<th valign="middle" align="center">Study area</th>
<th valign="middle" align="center">Crop type</th>
<th valign="middle" align="center">Number of samples</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center" rowspan="21">I</td>
<td valign="middle" align="center">Seed Maize</td>
<td valign="middle" align="center">36771</td>
<td valign="middle" align="center" rowspan="22">II</td>
<td valign="middle" align="center">Middle Rice</td>
<td valign="middle" align="center">27826</td>
<td valign="middle" align="center" rowspan="22">III</td>
<td valign="middle" align="center">Cotton</td>
<td valign="middle" align="center">301178</td>
</tr>
<tr>
<td valign="middle" align="center">Spring Maize</td>
<td valign="middle" align="center">25136</td>
<td valign="middle" align="center">Winter Wheat</td>
<td valign="middle" align="center">8295</td>
<td valign="middle" align="center">Spring Maize</td>
<td valign="middle" align="center">78567</td>
</tr>
<tr>
<td valign="middle" align="center">Greenhouse</td>
<td valign="middle" align="center">7659</td>
<td valign="middle" align="center">Spring Wheat</td>
<td valign="middle" align="center">97</td>
<td valign="middle" align="center">Seed Maize</td>
<td valign="middle" align="center">77360</td>
</tr>
<tr>
<td valign="middle" align="center">Woodland</td>
<td valign="middle" align="center">4610</td>
<td valign="middle" align="center">Spring Maize</td>
<td valign="middle" align="center">221613</td>
<td valign="middle" align="center">Grape</td>
<td valign="middle" align="center">36779</td>
</tr>
<tr>
<td valign="middle" align="center">Alfalfa</td>
<td valign="middle" align="center">3390</td>
<td valign="middle" align="center">Seed Maize</td>
<td valign="middle" align="center">61720</td>
<td valign="middle" align="center">Tomato</td>
<td valign="middle" align="center">27266</td>
</tr>
<tr>
<td valign="middle" align="center">Chili Pepper</td>
<td valign="middle" align="center">3014</td>
<td valign="middle" align="center">Sweet Potato</td>
<td valign="middle" align="center">4008</td>
<td valign="middle" align="center">Winter Wheat</td>
<td valign="middle" align="center">15119</td>
</tr>
<tr>
<td valign="middle" align="center">Onion</td>
<td valign="middle" align="center">2968</td>
<td valign="middle" align="center">Safflower</td>
<td valign="middle" align="center">1335</td>
<td valign="middle" align="center">Sunflower</td>
<td valign="middle" align="center">14561</td>
</tr>
<tr>
<td valign="middle" align="center">Winter Wheat</td>
<td valign="middle" align="center">2493</td>
<td valign="middle" align="center">Sunflower</td>
<td valign="middle" align="center">10084</td>
<td valign="middle" align="center">Gourd</td>
<td valign="middle" align="center">10917</td>
</tr>
<tr>
<td valign="middle" align="center">Bare Land</td>
<td valign="middle" align="center">2260</td>
<td valign="middle" align="center">Soybean</td>
<td valign="middle" align="center">9457</td>
<td valign="middle" align="center">Bara Land</td>
<td valign="middle" align="center">10691</td>
</tr>
<tr>
<td valign="middle" align="center">Grape</td>
<td valign="middle" align="center">1925</td>
<td valign="middle" align="center">Cotton</td>
<td valign="middle" align="center">6446</td>
<td valign="middle" align="center">Woodland</td>
<td valign="middle" align="center">8222</td>
</tr>
<tr>
<td valign="middle" align="center">Sorghum</td>
<td valign="middle" align="center">1796</td>
<td valign="middle" align="center">Sugar Beets</td>
<td valign="middle" align="center">10298</td>
<td valign="middle" align="center">Chili Pepper</td>
<td valign="middle" align="center">5939</td>
</tr>
<tr>
<td valign="middle" align="center">Stevia</td>
<td valign="middle" align="center">1653</td>
<td valign="middle" align="center">Stevia</td>
<td valign="middle" align="center">7753</td>
<td valign="middle" align="center">Silver Beet</td>
<td valign="middle" align="center">5457</td>
</tr>
<tr>
<td valign="middle" align="center">Sunflower</td>
<td valign="middle" align="center">1307</td>
<td valign="middle" align="center">Alfalfa</td>
<td valign="middle" align="center">6353</td>
<td valign="middle" align="center">Watermelon</td>
<td valign="middle" align="center">3027</td>
</tr>
<tr>
<td valign="middle" align="center">Spring Wheat</td>
<td valign="middle" align="center">1189</td>
<td valign="middle" align="center">Chili Pepper</td>
<td valign="middle" align="center">2007</td>
<td valign="middle" align="center">Sweet Potato</td>
<td valign="middle" align="center">2252</td>
</tr>
<tr>
<td valign="middle" align="center">Pear</td>
<td valign="middle" align="center">619</td>
<td valign="middle" align="center">Gourd</td>
<td valign="middle" align="center">1319</td>
<td valign="middle" align="center">Greenhouse</td>
<td valign="middle" align="center">1920</td>
</tr>
<tr>
<td valign="middle" align="center">Sweet Potato</td>
<td valign="middle" align="center">308</td>
<td valign="middle" align="center">Watermelon</td>
<td valign="middle" align="center">950</td>
<td valign="middle" align="center">Hops</td>
<td valign="middle" align="center">1698</td>
</tr>
<tr>
<td valign="middle" align="center">Sugar Beets</td>
<td valign="middle" align="center">223</td>
<td valign="middle" align="center">Greenhouse</td>
<td valign="middle" align="center">1641</td>
<td valign="middle" align="center">Nursery</td>
<td valign="middle" align="center">1636</td>
</tr>
<tr>
<td valign="middle" align="center">Soybean</td>
<td valign="middle" align="center">186</td>
<td valign="middle" align="center">Grape</td>
<td valign="middle" align="center">2430</td>
<td valign="middle" align="center">Spring Wheat</td>
<td valign="middle" align="center">1248</td>
</tr>
<tr>
<td valign="middle" align="center">Gourd</td>
<td valign="middle" align="center">142</td>
<td valign="middle" align="center">Indian Jujube</td>
<td valign="middle" align="center">2777</td>
<td valign="middle" align="center">Alfalfa</td>
<td valign="middle" align="center">805</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">Woodland</td>
<td valign="middle" align="center">13822</td>
<td valign="middle" align="center">Pumpkin</td>
<td valign="middle" align="center">381</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">Nursery</td>
<td valign="middle" align="center">5258</td>
<td valign="middle" align="center">Potato</td>
<td valign="middle" align="center">183</td>
</tr>
<tr>
<td valign="middle" colspan="2" align="center">Total Number of Samples</td>
<td valign="middle" align="center">97649</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">405489</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">605326</td>
</tr>
<tr>
<th valign="middle" align="center">Study area</th>
<th valign="middle" align="center">Crop type</th>
<th valign="top" colspan="2" align="center">Number of samples</th>
<th valign="top" colspan="2" align="center">Study area</th>
<th valign="top" colspan="2" align="center">Crop Type</th>
<th valign="top" align="center">Number of samples</th>
</tr>
<tr>
<td valign="middle" align="center" rowspan="4">IV</td>
<td valign="middle" align="center">Soybean</td>
<td valign="top" colspan="2" align="center">1207</td>
<td valign="middle" colspan="2" rowspan="5" align="center">V</td>
<td valign="top" colspan="2" align="center">Soybean</td>
<td valign="top" align="center">571</td>
</tr>
<tr>
<td valign="middle" align="center">Spring Maize</td>
<td valign="top" colspan="2" align="center">440</td>
<td valign="top" colspan="2" align="center">Spring Maize</td>
<td valign="top" align="center">515</td>
</tr>
<tr>
<td valign="middle" align="center">Middle Rice</td>
<td valign="top" colspan="2" align="center">39</td>
<td valign="top" colspan="2" align="center">Middle Rice</td>
<td valign="top" align="center">416</td>
</tr>
<tr>
<td valign="middle" align="center">Other</td>
<td valign="top" colspan="2" align="center">364</td>
<td valign="top" colspan="2" align="center">Other</td>
<td valign="top" align="center">197</td>
</tr>
<tr>
<td valign="middle" colspan="2" align="center">Total Number of Samples</td>
<td valign="bottom" colspan="2" align="center">2050</td>
<td valign="bottom" colspan="2" align="center"/>
<td valign="bottom" align="center">1699</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Methods</title>
<sec id="s4_1">
<label>4.1</label>
<title>Motivation</title>
<p>In this paper, we propose a new combined Transformer and Convolutional network structure, the new structure takes the Transformer as the main structure and is used for crop classification, thus naming the new structure a Cropformer.</p>
<p>Both the single Convolutional structure and the single Transformer structure have certain drawbacks for crop classification. Convolution focuses too much on the key issues local to the sequence and ignores the dependencies between long sequences (<xref ref-type="bibr" rid="B55">Vaswani et&#xa0;al., 2017</xref>). Although Transformer can learn the dependencies between long sequences, those of local information are insensitive (<xref ref-type="bibr" rid="B28">Liu et&#xa0;al., 2022</xref>). Fusing the two to achieve a complementary effect can improve the adaptability of the model to cope with crop classification in different scenarios. The new network structure uses an embedded structure (<xref ref-type="bibr" rid="B9">Devlin et&#xa0;al., 2018</xref>), as shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>, to embed the Convolutional part downstream of the multi-headed attention of the Transformer, so that the features with weights acquired by Transformer are input to the convolutional part, which can also be regarded as adding weight to the input of the Convolution, and thus the Convolution can pay more attention to those local features with larger weights. The Convolutional structure uses a new double-start residual connection, as shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2C</bold>
</xref>. This connection ensures that the original input with weights can be fed directly to the Feed Forward Layer without losing the information in the global features due to the addition of Convolutional modules, thus allowing the global information extracted by Transformer to be fused with the local information extracted by Convolution. We use a point-depth convolution structure (<xref ref-type="bibr" rid="B17">Hua et&#xa0;al., 2018</xref>) to solve the problem that adding convolution significantly increases the number of parameters in Convolution part, and we use a residual structure to allow the model to converge faster (<xref ref-type="bibr" rid="B16">He et&#xa0;al., 2016</xref>). In addition, we use two parameter-adjustable activation functions, GLU and Swish (<xref ref-type="bibr" rid="B8">Dauphin et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B42">Ramachandran et&#xa0;al., 2017</xref>), which ensure that the convolution structure can automatically select the best activation function for training based on the prior self-attention output, and both activation functions have the advantage of fast convergence, so that the whole model can be trained and converged more easily. Single convolutional structure and single Transformer structure have been very common in the field of crop classification, but methods combining these two structures are still less common in the field of crop classification. Therefore, building a classification model dominated by convolutional and Transformer structures is valuable in the field of crop classification.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Cropformer Block, Convolution Module and Crop Residual detailed architecture. In <bold>(A)</bold> MHAL extracts global information, CM extracts local information, FFN will get information further enhanced, and LN is used for layer normalization to prevent model overfitting. <bold>(B)</bold> shows the two-start residual connection, and dropout is used to speed up the model training and prevent overfitting. GLU and Swish are the two learnable activation functions in <bold>(C)</bold>, and BN is used for batch normalization to prevent model overfitting.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g002.tif"/>
</fig>
<p>Irregular time series are difficult to apply directly due to their irregularity and to solve this problem, we introduced positional encoding in Cropformer. The effectiveness of positional encoding in the field of natural language processing(NLP) has been demonstrated (<xref ref-type="bibr" rid="B9">Devlin et&#xa0;al., 2018</xref>), but there is no precedent in the field of crop classification as a method for solving irregular time series. Specifically, we encode the observation points of the acquired valid remote sensing images to form temporal features, which are fused with the spectral features and input into the model together. The model will learn the spectral features at the corresponding positions according to the temporal features, and no misalignment of spectral values will occur due to different sequence lengths.</p>
<p>Labeled sample data is costly to obtain, while unlabeled remote sensing image data is easily available, and using unlabeled data to improve the learning ability of the model is competitive compared to other models (<xref ref-type="bibr" rid="B61">Yuan and Lin, 2021</xref>; <xref ref-type="bibr" rid="B62">Yuan et&#xa0;al., 2022</xref>). Pre-training using unlabeled data in this study forces the model to learn the crop growth patterns from unlabeled data, thus accumulating a large amount of prior knowledge. Specifically, we randomly add noise to some of the nodes in the sequence of unlabeled data and pre-train the model using a self-supervised training approach, thus allowing the model to learn the spatial-temporal relationships of crops at different time nodes. This improves the generalization ability of the model as well as provides prior knowledge for supervised classification.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Cropformer</title>
<p>The Cropformer architecture consists of three parts: Token Embedding (TE), Position Embedding (PE), and Cropformer Block (CB), whose architecture is shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. TE is a Linear Layer that projections the spectral sequence into a sequence feature vector of dimension <italic>d</italic>, i.e., equation (1). PE encodes the time series into a temporal feature vector of dimension <italic>d</italic> by equation (2). The sequence feature vector and the temporal feature vector are concatenated into a new vector as the input of CB.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Cropformer architecture. Where Token Embedding denotes encoding of spectral information in time series and Position Embedding denotes encoding of temporal information in time series. Cropformer Block details can be shown by <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, <xref ref-type="fig" rid="f4">
<bold>4</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g003.tif"/>
</fig>
<disp-formula>
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>sin</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>1000</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#xa0;&#xa0;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>cos</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>1000</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#xa0;&#xa0;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>Concat</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>i</italic>&#x2208;[0,<italic>N</italic>]<italic>,k</italic>&#x2208;[0,<italic>d</italic>] , <italic>N</italic> denotes the sequence (time series) length and <italic>doy</italic> represents the difference of valid sampling points. Encoding <italic>doy</italic> ensures that each growth node of the crop has a fixed temporal feature vector corresponding to it, so the input sequence can be irregular. Although irregular inputs can improve image utilization, they can lead to a very limited number of effective images acquired when subjected to practical conditions such as cloud occlusion, which requires the model to be able to learn key features from the constrained inputs.</p>
<p>CB in Cropformer can exist <italic>N</italic>, forming a network structure with depth <italic>N</italic>. However, there is a limit to the size of <italic>N</italic>, and infinite increase does not significantly improve the results, and its structure is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>. The CB consists of three important components: Multi-Head Attention Layer (MHAL), Convolution Module (CM), and Feed-Forward Network (FFN). MHAL is good at capturing global information (<xref ref-type="bibr" rid="B61">Yuan and Lin, 2021</xref>), while CM can effectively use local features, and combining the two can achieve more comprehensive learning of crop growth and development.</p>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref> shows that Single MHAL architecture. MHAL takes the joint vector of sequence + doy as input, by performing three linear projections on the input vector. The outputs of the first two Linear Layer are selected for scaled dot product and the fraction of each feature is calculated using Softmax, i.e., Equation. (4). The obtained feature scores are dotted multiplied once more with the output of the third projection, and finally the output of MHAL is obtained. MHAL is to calculate the feature scores between different positions of sequences, it learns the dependencies between sequences and obtains the global sequence information. The weight of the important sequence information in this part is scaled up, making more emphasis on this part in the CM part.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>
<bold>(A)</bold> Single MHAL architecture <bold>(B)</bold> FFN architecture. The inputs in <bold>(A)</bold> are the visualization results of the time series. L1-3 denote the three Linear Layers, but their weights are not the same.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g004.tif"/>
</fig>
<disp-formula>
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>max</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mi>d</mml:mi>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>L<sub>j</sub>
</italic> represents <italic>j</italic>-th Linear Layer in MHAL; <italic>d</italic> represents the feature vector of dimension; <italic>x<sub>i</sub>
</italic>denotes model input, and <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes MHAL output. Softmax is used to calculate the score of each feature.</p>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref> shows that FFN architecture. FFN is used to enhance the expressiveness of the features and consists of two linear layers and an activation function.</p>
<p>CM is composed of Layer Normalization (LN) (<xref ref-type="bibr" rid="B3">Ba et&#xa0;al., 2016</xref>), Crop Residual (CR), and Dropout (<xref ref-type="bibr" rid="B51">Srivastava et&#xa0;al., 2014</xref>), whose structure is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>. The mathematical expression of the CB part is</p>
<disp-formula>
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>M</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where <italic>LN</italic> represents Layer Normalization, <italic>CM</italic> represents Convolution Module, and <italic>FFN</italic> represents Feed-Forward Network; <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes CM input, <italic>x</italic>
<sup>&#x2032;</sup> denotes <italic>FFN</italic> input, and y<italic>
<sub>i</sub>
</italic> denotes model input.</p>
<p>CR is a pure Convolutional structure, we use a new double-start Shortcut connection that can fuse two kinds of features, which can guarantee lossless fusion of global and local features. The depth-separable convolution layer is used in CR, which effectively reduces the number of parameters and ensures that the model can be trained faster. Two activation functions, GLU and Swish, are used in the Pointwise Convolution Layer and Depth Convolution Layer, respectively, and in the last Batch Normalization (BN) (<xref ref-type="bibr" rid="B20">Ioffe and Szegedy, 2015</xref>), which is more suitable for convolution operations, is added to one convolution layer, and its architecture is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2C</bold>
</xref>.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Crop classification framework</title>
<p>The crop classification framework using Cropformer is divided into a pre-training part and a fine-tuning part, as shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. In the pre-training part, we use an SSL training approach for predicting missing values (<xref ref-type="bibr" rid="B9">Devlin et&#xa0;al., 2018</xref>), and it is worth noting that this part is trained entirely with unlabeled data. In the input continuous irregular time series, we randomize the time series of the MASK part of the input sequence sampling points, and Cropformer predicts the value of the MASK part by learning the spatial-temporal contextual relationships between the sequences, which allows the model to fully learn each node of crop growth and development. The loss function of Cropformer uses the Mean-Square Error (MSE) between the original and predicted sequences, i.e.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Cropformer crop classification framework. <italic>Seq</italic> represents the input time series, <italic>Mask</italic> represents the time series being masked, and <italic>Doy</italic> represents the sampling time point. The fine-tuning phase has the same network structure as the pre-training phase, but the fine-tuning phase has a Classification module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g005.tif"/>
</fig>
<disp-formula>
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mover accent="true">
<mml:mi>e</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mo>;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>seq</italic> represents the sequence value of MASK, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mover accent="true">
<mml:mi>e</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the predicted sequence value and <italic>N</italic> denotes the number of masked sequence values. When the predicted values are infinitely close to the true values, it shows that Cropformer already can recognize the growth and development patterns of crops, thanks to pre-training using a large amount of unlabeled data. Although accurate crop types cannot be obtained by pre-training with unlabeled data, the growth patterns of a large number of crops expressed as time series are learned. These learned crop growth patterns are passed on to the fine-tuning stage as prior knowledge, thus ensuring that the best classification results can be obtained more efficiently in the fine-tuning stage.</p>
<p>After completing the pre-training, the resulting parameters from the pre-training part are transferred to the fine-tuning part for use. Because of the effectiveness of the pre-training part, the fine-tuning part does not take much time. Also, in the network structure part, only a simple classification module is added behind the Cropformer and then a supervised fine-tuning process is performed.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Experiment design and settings</title>
<p>We had applied Cropformer in several crop classification scenarios to demonstrate its generality namely (1) Full-season crop classification; (2) In-season crop classification; (3) Few-sample crop classification, (4) Spatial transfer of classification model. The experimental scheme is shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>. Specifically, (1) The classification ability of Cropformer was tested using current-year unlabeled data for pre-training and the full current-year field sample for fine-tuning, and compared to the best existing approaches RF (<xref ref-type="bibr" rid="B39">Pelletier et&#xa0;al., 2016</xref>), Res-18 (<xref ref-type="bibr" rid="B54">Thenmozhi and Reddy, 2019</xref>), SIFT-BERT (<xref ref-type="bibr" rid="B61">Yuan and Lin, 2021</xref>), Performer (<xref ref-type="bibr" rid="B6">Choromanski et al., 2020</xref>) and ALBERT (<xref ref-type="bibr" rid="B25">Lan et&#xa0;al., 2019</xref>); (2) In the in-season crop classification experiments, historical data were used as a pre-training data source, and the field sampling data in the current year were divided by month as fine-tuned data, e.g., March-end of April for the first stage of early detection and March-end of May for the second stage, with the input time series gradually becoming longer until all-time series were included; (3) In the few-sample crop classification experiment, two few-sample scenarios were simulated as balanced and unbalanced crop distributions. We designed experiments with 1% of labeled samples drawn from each class and a fixed number of labeled samples drawn from each class (the number of samples with the lowest number of all classes as the number of draws) as the fine-tuning training dataset; (4) Transfer learning can solve the problem of insufficient labeled samples in the target domain, and we set up two transfer methods in this experiment. One is to transfer the model from training in a region with sufficient samples to the target region, and the other was to transfer the model from training in a geographically similar region to the target region.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>The experimental process uses historical data and current year data as pre-training data. The acquisition time of image data is from March to October. The bands used are blue, green, red, and near-red. The red serial number represents the number of crop classification scenario.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g006.tif"/>
</fig>
<p>To evaluate the experimental results, besides the visually comparison of the classified crop maps, Overall Accuracy (OA), Average Accuracy (AA), and F1 scores were used to quantify the classification performance of different methods. In addition, all the results were the average of the three experimental results.</p>
<p>The hyperparameter settings of Cropformer are divided into two parts: network structure and training optimization, both of which are closely related to the performance of the model. For the network structure, the number of CB is set to 3 and the number of CR is set to 2. The number of Head in MHAL is set to 8 and the dimension of Linear Layers is set to 256; the dimension of Linear Layers in FFN is set to 1024; the size of convolutional kernels in CR is set to 7&#xd7;7, padding is 3, stride is 1. The number of channels in both Pointwise Conv and The number of channels of both Pointwise Conv and Depthwise Conv is 256. For the training optimization, the pre-training stage was performed with 200 epochs, batch size set to 512, set to initial learning rate set to 1e-4, decay after 10 epochs, dropout set to 0.1, and the optimizer is selected as Adam; the fine-tuning stage was performed with 10 epochs, batch size set to 256, learning rate set to 1e-5, and dropout set to 0.1. The input size of the model can be obtained from equation (7)</p>
<disp-formula>
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>*</mml:mo>
<mml:mi>b</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>t_num</italic> denotes the number of images acquired which is the length of the time series, and <italic>band_num</italic> denotes the number of bands. The number of bands in this paper is 4, but the inputs to the model in this paper are diverse because the inputs used are irregular time series.</p>
<p>The entire experiment was run on a Windows platform configured with an i7-11700 K @ 3.60 GHz, 32 G RAM, and NVIDIA GeForce RTX 3080 GPU (10 GB RAM), and all programs were written using the python language.</p>
</sec>
</sec>
<sec id="s5" sec-type="results">
<label>5</label>
<title>Results and analysis</title>
<sec id="s5_1">
<label>5.1</label>
<title>Full-season crop classification</title>
<p>In our experiments, we compared Cropformer with five competing methods. RF had advantages in processing sequential data and hence was used as a baseline for traditional machine learning algorithms. The advanced deep learning methods Res_18, SIFT_BERT, Performer, and ALBERT were used as comparisons and they showed good results in crop classification as well as sequence data processing. Among them, the input of Res_18 was a three-dimensional tensor, the input of RF and Performer was a regular time series, and the input of SIFT_BERT and ALBERT were irregular time series in line with the input of Cropformer, and the experimental results were shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance comparison of cropformer and other classifiers in three study areas.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Study<break/>Area</th>
<th valign="top" align="center">Methods</th>
<th valign="top" align="center">RF</th>
<th valign="top" align="center">Res-18</th>
<th valign="top" align="center">Performer</th>
<th valign="top" align="center">SIFT_BERT</th>
<th valign="top" align="center">ALBERT</th>
<th valign="top" align="center">Cropformer</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="2" align="center">Study<break/>Area I</td>
<td valign="top" align="center">AA(%)</td>
<td valign="top" align="center">56.43</td>
<td valign="top" align="center">50.14</td>
<td valign="top" align="center">54.95</td>
<td valign="top" align="center">71.11</td>
<td valign="top" align="center">70.66</td>
<td valign="top" align="center">
<bold>73.16</bold>
</td>
</tr>
<tr>
<td valign="top" align="center">OA(%)</td>
<td valign="top" align="center">80.72</td>
<td valign="top" align="center">73.08</td>
<td valign="top" align="center">74.10</td>
<td valign="top" align="center">79.65</td>
<td valign="top" align="center">78.58</td>
<td valign="top" align="center">
<bold>81.93</bold>
</td>
</tr>
<tr>
<td valign="top" rowspan="2" align="center">Study<break/>Area II</td>
<td valign="top" align="center">AA(%)</td>
<td valign="top" align="center">48.16</td>
<td valign="top" align="center">34.73</td>
<td valign="top" align="center">60.33</td>
<td valign="top" align="center">63.62</td>
<td valign="top" align="center">63.31</td>
<td valign="top" align="center">
<bold>64.81</bold>
</td>
</tr>
<tr>
<td valign="top" align="center">OA(%)</td>
<td valign="top" align="center">85.82</td>
<td valign="top" align="center">84.80</td>
<td valign="top" align="center">84.65</td>
<td valign="top" align="center">84.56</td>
<td valign="top" align="center">85.39</td>
<td valign="top" align="center">
<bold>86.32</bold>
</td>
</tr>
<tr>
<td valign="top" rowspan="2" align="center">Study<break/>Area III</td>
<td valign="top" align="center">AA(%)</td>
<td valign="top" align="center">60.02</td>
<td valign="top" align="center">43.85</td>
<td valign="top" align="center">58.89</td>
<td valign="top" align="center">57.42</td>
<td valign="top" align="center">
<bold>62.37</bold>
</td>
<td valign="top" align="center">61.23</td>
</tr>
<tr>
<td valign="top" align="center">OA(%)</td>
<td valign="top" align="center">85.61</td>
<td valign="top" align="center">78.29</td>
<td valign="top" align="center">83.69</td>
<td valign="top" align="center">83.97</td>
<td valign="top" align="center">
<bold>85.99</bold>
</td>
<td valign="top" align="center">85.70</td>
</tr>
<tr>
<td valign="top" rowspan="2" align="center">Study<break/>Area IV</td>
<td valign="top" align="center">AA(%)</td>
<td valign="top" align="center">73.26</td>
<td valign="top" align="center">59.71</td>
<td valign="top" align="center">78.10</td>
<td valign="top" align="center">
<bold>82.91</bold>
</td>
<td valign="top" align="center">79.66</td>
<td valign="top" align="center">81.52</td>
</tr>
<tr>
<td valign="top" align="center">OA(%)</td>
<td valign="top" align="center">79.71</td>
<td valign="top" align="center">71.71</td>
<td valign="top" align="center">81.46</td>
<td valign="top" align="center">82.44</td>
<td valign="top" align="center">83.49</td>
<td valign="top" align="center">
<bold>84.39</bold>
</td>
</tr>
<tr>
<td valign="top" rowspan="2" align="center">Study<break/>Area V</td>
<td valign="top" align="center">AA(%)</td>
<td valign="top" align="center">53.92</td>
<td valign="top" align="center">48.45</td>
<td valign="top" align="center">60.86</td>
<td valign="top" align="center">
<bold>72.42</bold>
</td>
<td valign="top" align="center">70.92</td>
<td valign="top" align="center">70.23</td>
</tr>
<tr>
<td valign="top" align="center">OA(%)</td>
<td valign="top" align="center">71.79</td>
<td valign="top" align="center">71.49</td>
<td valign="top" align="center">71.06</td>
<td valign="top" align="center">77.87</td>
<td valign="top" align="center">78.91</td>
<td valign="top" align="center">
<bold>79.15</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bolded indicates best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>
<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> showed that Cropformer obtained OA of 81.93%, 86.32%, 85.70%, 84.39%, and 79.15% in the five study areas, respectively. Compared to the baseline RF of the traditional method, the increase was more than 5% in study areas IV-V with fewer samples, while in study areas I-III with abundant samples, the increase was not significant. This demonstrated the saturation of the classification accuracy achieved by the classification algorithm when samples were sufficient. In study area IV, Cropformer obtained AA of 81.52%, which was very close to the OA (84.39%) obtained in this region, while in study areas II and III it obtained AA of 64.81% and 61.23%, respectively, which was more than 20% different from the OA (86.32%, 85.70%) obtained, and this difference was more pronounced in RF (37%, 25%). This was due to the extremely unbalanced distribution of samples in study area II and III, where the sample size of maize was more than half of the total sample size, causing the AA to be insignificant, which is consistent with the findings of (<xref ref-type="bibr" rid="B57">Wang et&#xa0;al., 2022</xref>). Sample imbalance can bias the classification results toward the more numerous categories. The results showed that Cropformer in full-season crop mapping enabled to achieve better and more stable classification results even in situations where the sample conditions were not ideal, while the performance of the traditional classification method (RF) was not stable and vulnerable to realistic conditions, which is in line with the conclusions of (<xref ref-type="bibr" rid="B59">Xu et&#xa0;al., 2020</xref>).</p>
<p>SIFT-BERT and ALBERT outperformed Cropformer for classification in a few cases. ALBERT achieved AA/OA of 62.37%/85.99% respectively, which was better than Cropformer (61.23%/85.70%); SIFT-BERT achieved AA of 82.91%/72.42% in study areas IV-V, again better than Cropformer (81.52%/70.23%). However, in most cases Cropformer had a clear advantage in both AA and OA. In study areas II, III, V, where the number of valid images was low (valid images of 27/24/12, respectively), Cropformer had a 1%-8% improvement in OA compared to RF and Performer (22 valid images) using regular time series, while the improvement in AA was very significant (3%-16%). This indicated that Cropformer is able to learn more useful features from a finite length sequence. In fact, regular time series required resampling operation, and in the case of limited number of images, this operation would destroy the original information of the time series and thus had an impact on the classification results. Res_18 performed the worst among all methods, obtaining only OA of 84.80% in study area II, and no more than AA/OA of 60%/80% in other study areas, which was an unacceptable result. Although the Res_18 used a more informative three-dimensional tensor as input, its classification accuracy was not outstanding. Therefore, the direct use of one-dimensional time series as input results as well as efficiency would be more advantageous.</p>
<p>
<xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7B, C</bold>
</xref> showed that in the case of sufficient samples, all three methods had a significant improvement in accuracy after pre-training, among which ALBERT had the most significant improvement (5%), while Cropformer and SIFT-BERT do not have a significant increase (2%-3%). The accuracy improvement of the three methods after pre-training was very obvious, especially in study areas IV-V, where the sample size was very small, and the highest improvement was up to 14.89%. The results showed that pre-training not only improves the crop classification accuracy, but also effectively reduced the model&#x2019;s demand for samples, which was consistent with the findings of (<xref ref-type="bibr" rid="B61">Yuan and Lin, 2021</xref>; <xref ref-type="bibr" rid="B62">Yuan et&#xa0;al., 2022</xref>). In addition, the the average improvement in the five study areas of the Cropformer (6.95%) after pre-training was higher than that of BERT (3.22%) and ALBERT (5.2%), which indicated that the Convolution-Transformer structure in the Cropformer had better learning ability compared to the single-structured Transformer. By focusing on both global and local crop growth patterns, we could not only focus on the key growth nodes of crops but also capture the dynamics of crops throughout the reproductive period, and combine two important discriminatory approaches to better distinguish between different types of crops.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Performance comparison of Cropformer and other classifiers in five study area.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g007.tif"/>
</fig>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>In-season crop classification</title>
<p>Historical unlabeled data were selected as pre-training data in the in-season crop classification experiment. We believed that crop growth information could be learned by training on historical data, even if it was not from the current year. Therefore, we used historical data as pre-training data in in-season crop classification and irregular time series of different lengths of the current year as fine-tuning data. The classification time was a one-month interval, with the end of April as the start time and the end of September as the end time, and the experimental results were shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Early detection results with time transformation in the five study areas. The first row of which is OA and the second row is AA for the period from the end of April to the end of September. <bold>(A)</bold> OA for in-season crop classification <bold>(B)</bold> AA for in-season crop classification.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g008.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8A</bold>
</xref> showed Cropformer dominance in early crop growth (end of April - end of May), especially in study area I, III, V where OA of 72.74%, 77.47% and 68.09% were obtained, while RF only obtained 70.68%, 74.71% and 65.81%. However, the advantage of Cropformer was not obvious in the middle of the crop growth period (end of June - end of August), especially in study area II and study area V, where the number of valid images for this time period was extremely low, thus leading to the inability to obtain valid classification results using the irregular time series method. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8B</bold>
</xref> showed that the advantage of Cropformer in evaluating the in-season crop classification with AA Cropformer can improve 1%-6% in early crop growth (end of May) compared to other methods, while in mid-growth the performance was comparable to SIFT-BERT and ALBERT, but significantly better than RF and Performer, which was most evident in study area I and study area II, where the improvement was close to 10%. Comparing <xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8A, B</bold>
</xref>, it can be found that in in-season crop classification, although RF can achieve some advantage (when evaluated with OA), it was the classification of a large number of teste data into a larger number of categories that achieved better results, so that the classification results are no longer advantageous when evaluated with AA. The pre-training of the accumulated prior knowledge improved the ability of the model to detect different classes of crops, so the model with pre-training can be applied to areas with uneven sample distribution and complex crop types. In addition, the pre-trained data were derived from historical data, which would make the historical data as pre-trained data had an impact on the classification accuracy if the historical data were different from the current year&#x2019;s data in terms of sowing time and other agricultural activities, resulting in some differences in crop growth stages from the current year.</p>
<p>The earlier the crops were classified, the more important the impact on agricultural production, so we further investigated the earliest point in time of the year when different crops were classified using Cropformer in three areas rich in crop types. Since different crops had different sowing times, the earliest time that could be classified would be different and similar. We analyzed the end of April and the end of July as two important points in the in-season crop classification, and the experimental results were shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>F1 scores for each crop type in the three study areas using Cropformer at the end of April, July and September. Where the green axis headings indicate vegetables and the red axis headings indicate fruits. <bold>(A)</bold> Hexi Corridor F1 Score <bold>(B)</bold> Ili River Valley F1 Score <bold>(C)</bold> Tianshan Corridor F1 Score.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g009.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> shows for half of the crops in all three study areas, more than 50% of the F1 scores were available at the end of April and nearly 60% of the crops had more than 70% of the F1 scores at the end of July, with a higher percentage in the Hexi Corridor(80% of the crop had an F1 score above 50% at the end of April and 70% of the crop had an F1 score above 70% at the end of July), due to the relatively balanced distribution of samples in the Hexi Corridor compared to the other two study areas. The unbalanced sample distribution was very unfair for crops with relatively small sample sizes, such as Chili Pepper and Gourd in the Ili River Valley, which were very susceptible to confounding with other crops (both had relatively low F1 scores in the three study areas), and because of the small number of available training samples, a large amount of confounding could occur. In all study areas, the F1 scores for each crop type at the end of July were very close to the F1 scores at the end of September, and even some crops had higher F1 scores at the end of July than at the end of September. When the crops were close to maturity or harvest, the time series information of crops was very close at this time, and if we continue to add time-series information, it would generate redundant information or useless information, which would make the model misclassify, so we could consider the end of July as a better time point for early classification using Cropformer. <xref ref-type="bibr" rid="B15">Hao et&#xa0;al. (2018)</xref>also proved the conclusion that accuracy consistent with crop maturity can be obtained in July-August. Cropformer uses irregular sequences and acquisition of very limited spectral information about crops but still allows crop identification at an earlier time, which shows that Cropformer can be applied to the classification of in-season crops in regions with complex crop types.</p>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Few-sample crop classification</title>
<p>In the few-sample crop classification experiments, we modeled two schemes to fit the few-sample scenario, i.e., using only 1% labeled samples per category and using only a fixed number of labeled samples per category (the fixed number referred to the number of samples with the least amount of sample size among all categories). We selected RF, SIFT-BERT, and ALBERT, which showed excellent results in previous experiments, as a comparison and conducted experiments in three areas with rich crop types. Each experiment was randomly selected three times to take the average value as the result and the experimental results were shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Comparison of results for different proportions of labeled samples in the three study areas. <bold>(A)</bold> OA for few-sample crop classification <bold>(B)</bold> AA for few-sample crop classification.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g010.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10A</bold>
</xref> showed Cropformer achieved OA of 76.27%, 85.18%, and 85.40% in the three study areas when using only 1% labeled samples, which was only 5.66%, 1.14%, and 0.3% less than using all labeled samples. When trained using the 1% sample, RF showed a significant OA decrease of 10.4%, 11.47%, and 8.22% in the three study areas, respectively, while the OA decrease using pre-trained SIFT-BERT, ALBERT, and Cropformer was not significant and only showed a significant OA decrease in study area I. This further demonstrated the effectiveness of pre-training. Moreover, Cropformer also performed optimally when only 1% of the samples were used, but ALBERT achieved comparable performance to Cropformer, while SIFT-BERT showed a significant drop in performance compared to the previous performance. When using the minimum number of samples per class, the accuracy of all methods decreased, which was due to the uniform trend and extremely reduced number of samples in the training set, but the distribution in the test set was still extremely uneven, which leaded to a significant decrease in accuracy.</p>
<p>
<xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10B</bold>
</xref> showed that the AA of Cropformer reached 74.03%,68.16% and 66.72% when using the minimum number of samples per class, which was 0.87%, 3.35% and 5.49% higher than the AA using the full sample fine-tuning. This further supported the previous conclusion that when the sample distribution was uneven, the category with the larger sample size dominates. Whether using the minimum number of samples per class or 1% samples per class, classification results of Cropformer still outperformed other methods, while SIFT-BERT and ALBERT were second. As with the previous classification results evaluated in terms of AA, the AA of RF was the lowest among all methods, which showed that RF was not applicable to samples with unbalanced distribution. Overall, Cropformer achieved competitive classification results in few-sample context classification, both in unbalanced and balanced samples.</p>
</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Spatial transfer of classification model</title>
<p>If the target region was not rich in labeled samples for training, transferring the model trained in the source region to the target region could reduce the problem of sparse labeled samples. We set up two types of transfer across regions, one was between regions with similar geographic and climatic conditions(TL1), and the other was to use regions with rich labeled samples to transfer to regions with fewer labeled samples(TL2). Study area II was the area with abundant labeling samples, and study area IV and study area V had similar geographical and climatic conditions. Since study area II and study area V were not in the same geographical area, we selected pre-training data from the two areas for pre-training. Where Pre_W and Pre_E represented the pre-training sets for the Northwest and Northeast regions of China, respectively. The crops were common to all three areas: soybean, spring maize, middle rice, and others. The results of the experiment are shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Comparing the results of different transfer strategies.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Pre-training Area</th>
<th valign="top" align="center">Methods</th>
<th valign="top" colspan="2" align="center">Study AreaIV&#x2192; Study AreaV</th>
<th valign="top" colspan="2" align="center">Study AreaII&#x2192; Study AreaV</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">AA(%)</td>
<td valign="top" align="center">OA(%)</td>
<td valign="top" align="center">AA(%)</td>
<td valign="top" align="center">OA(%)</td>
</tr>
<tr>
<td valign="top" rowspan="2" align="center">No_Pre</td>
<td valign="top" align="center">RF</td>
<td valign="top" align="center">37.93</td>
<td valign="top" align="center">25.21</td>
<td valign="top" align="center">41.52</td>
<td valign="top" align="center">47.68</td>
</tr>
<tr>
<td valign="top" align="center">Performer</td>
<td valign="top" align="center">33.65</td>
<td valign="top" align="center">30.54</td>
<td valign="top" align="center">43.62</td>
<td valign="top" align="center">44.38</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">Cropformer</td>
<td valign="top" align="center">25.96</td>
<td valign="top" align="center">62.13</td>
<td valign="top" align="center">27.68</td>
<td valign="top" align="center">62.98</td>
</tr>
<tr>
<td valign="top" rowspan="3" align="center">Pre_W</td>
<td valign="top" align="center">ALBERT</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">22.65</td>
<td valign="top" align="center">43.67</td>
</tr>
<tr>
<td valign="top" align="center">BERT</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">23.34</td>
<td valign="top" align="center">32.77</td>
</tr>
<tr>
<td valign="top" align="center">Cropformer</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">25.65</td>
<td valign="top" align="center">61.70</td>
</tr>
<tr>
<td valign="top" rowspan="3" align="center">Pre_E</td>
<td valign="top" align="center">ALBERT</td>
<td valign="top" align="center">
<bold>35.93</bold>
</td>
<td valign="top" align="center">52.34</td>
<td valign="top" align="center">29.32</td>
<td valign="top" align="center">63.40</td>
</tr>
<tr>
<td valign="top" align="center">BERT</td>
<td valign="top" align="center">28.36</td>
<td valign="top" align="center">62.98</td>
<td valign="top" align="center">44.78</td>
<td valign="top" align="center">62.55</td>
</tr>
<tr>
<td valign="top" align="center">Cropformer</td>
<td valign="top" align="center">29.65</td>
<td valign="top" align="center">
<bold>63.77</bold>
</td>
<td valign="top" align="center">
<bold>45.46</bold>
</td>
<td valign="top" align="center">
<bold>64.26</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bolded indicates best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>
<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> showed that Cropformer achieved OA of 62.13% in TL1 when no pre-training was used, which was a 36.92% and 31.59% improvement compared to RF and Performer. In TL2 the OA reached 62.98%, a 15.3% and 18.6% improvement compared to RF and Performer. However, the AA of Cropformer was not outstanding among the two transfer methods. When using Pre_W as pre-training, the OA of Cropformer reached 61.70%, which was still more than 20% improvement compared to ALBERT and SIFT-BERT. When using Pre_E as pre-training, Cropformer achieved OA of 63.77% in TL1 and OA of 64.26% in TL2. Compared to the first two pre-training methods, the OA improvement of Cropformer was much less using the third pre-training method, which indicated that the region of pre-trained data need to be consistent with the region of fine-tuned data. However, Cropformer can overcome the scenario of regional inconsistency, reflecting Cropformer&#x2019;s ability to transfer across regions. Cropformer&#x2019;s ability to capture key information about the crop and understand crop growth patterns from the entire reproductive period of the crop allowed Cropformer to identify crops in different regions faster and better, which was important reason for Cropformer&#x2019;s good spatial transfer capability.</p>
<p>In the no-pre-training scenario, the average OA/AA of the three methods in TL2 reached 51.68%/37.61%, which was an improvement of 12.48%/5.1%, respectively, compared to that in TL1 (39.2%/32.51%). When pre-training with Pre_E, the average OA/AA of the three methods in TL2 reached 63.40%/39.85%, which was an improvement of 3.7%/8.54% compared to TL1 (59.70%/31.31%), respectively. The results indicated that TL2 outperformed TL1, and thus it can be assumed that models trained in areas with rich label samples outperformed those trained in areas with similar geographical location and climate.</p>
</sec>
<sec id="s5_5">
<label>5.5</label>
<title>Processing efficiency</title>
<p>We compared the processing speed of Cropformer with the other three methods in the Ili River Valley and ensured that all experimental settings were consistent, and the experimental results were shown in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Comparison of processing speed between cropformer and other methods(s).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Methods</th>
<th valign="top" align="center">RF</th>
<th valign="top" align="center">Res-18</th>
<th valign="top" align="center">Performer</th>
<th valign="top" align="center">ALBERT</th>
<th valign="top" align="center">SIFT-BERT</th>
<th valign="top" align="center">Cropformer</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Data pre-processing</td>
<td valign="top" align="center">523.93</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">523.92</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">Pre-training epoch</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">230.36</td>
<td valign="top" align="center">275.25</td>
<td valign="top" align="center">315.88</td>
</tr>
<tr>
<td valign="top" align="left">Training epoch</td>
<td valign="top" align="center">271.65</td>
<td valign="top" align="center">1165.24</td>
<td valign="top" align="center">470.69</td>
<td valign="top" align="center">196.85</td>
<td valign="top" align="center">242.84</td>
<td valign="top" align="center">250.63</td>
</tr>
<tr>
<td valign="top" align="left">All time consuming</td>
<td valign="top" align="center">795.58</td>
<td valign="top" align="center">1165.24</td>
<td valign="top" align="center">994.61</td>
<td valign="top" align="center">427.21</td>
<td valign="top" align="center">518.09</td>
<td valign="top" align="center">566.51</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> showed that the total time consumption of Cropformer (566.51s) was higher than that of BERT (518.09s), ALBERT (427.21s), and Performer (994.61s), but lower than that of RF (795.58s) and Res-18 (1165.24s). The main time consumption of RF and Performer consumption lied in data preprocessing (523.93s), which accounted for 65.86% and 52.68% of all time consumed, which severely limited the efficiency of RF and Performer. Cropformer was more time consuming than SIFT-BERT and ALBERT because we added a convolutional part to the network structure. ALBERT had the advantage of having a very small number of parameters, which was an important reason why its efficiency is the best among all methods.</p>
</sec>
<sec id="s5_6">
<label>5.6</label>
<title>Comparison of the crop maps</title>
<p>We selected two 10&#xa0;km &#xd7;10 km areas in each of the five study areas for full-season mapping using Cropformer, and compared the results with those of RF, and SIFT-BERT, as shown in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>. Results of crop mapping by other methods can be viewed in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Map of crop distribution in selected areas of the five study areas. For each study area, from left to right, remote sensing images, RF mapping, SIFT- BERT mapping, and Cropformer mapping.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1130659-g011.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> showed the mapping differences between the three methods were more pronounced in the first three study areas with rich crop types, while the three methods were similar in the last two study areas with similar cropping structures. The most significant differences in the maps among the five methods were found in the Tianshan Corridor, for example, in the second 10&#xa0;km &#xd7; 10&#xa0;km area, Cropformer identified most of the crops as grapes, while RF identified most of the crops as spring maize and sunflower, and SIFT-BERT identified them as seeded mazie and cotton. Based on the field survey in Tianshan Corridor, an important grape production base in China, the prediction of Cropformer was more accurate. In the area with complex planting structures, Cropformer still showed good mapping performances. The parcel distribution of Cropformer mapping is very close to plot distribution of the original remote sensing image, and the mapping results of other methods had a serious pepper effect and look more fragmented. In the latter two study areas where the plot size was small and clustered distributed, Cropformer still overcame the pepper effect and there were few cases of fragmented plots. Overall Cropformer had good mapping results and provides a possible solution for large-scale remote sensing mapping.</p>
</sec>
</sec>
<sec id="s6" sec-type="discussion">
<label>6</label>
<title>Discussion</title>
<p>In this paper, we proposed a new crop classification method that can be applied to multi-scenario crop classification, and its generality and validity were demonstrated in five study areas with complex crop growing structures.</p>
<p>The success of Cropformer lied in the ability to focus on both global information about crop growth and to capture key features of crop growth to achieve a more comprehensive feature representation from the perspective of feature complementarity. For the features extracted by the model, neither local nor global features can fully characterize the crop growth pattern (<xref ref-type="bibr" rid="B13">Gulati et&#xa0;al., 2020</xref>). Therefore, better results were obtained using more comprehensive and integrated features. The role of pre-training was to help the model better understand the crop growth pattern, and when pre-training was introduced in the classification method, the classification accuracy was significantly improved, which was consistent with the findings in the literature (<xref ref-type="bibr" rid="B61">Yuan and Lin, 2021</xref>; <xref ref-type="bibr" rid="B62">Yuan et&#xa0;al., 2022</xref>). The implication behind pre-training was to make the model learn the contextual relationships of the time series by forcing the model to learn them through self-supervised training when inputting unlabeled time series, thus summarizing the time series patterns of the crop. These laws were used as prior knowledge to the fine-tuning phase, which both reduced the need for labeled samples and sped up the convergence of the model (<xref ref-type="bibr" rid="B27">Li et&#xa0;al., 2020</xref>). The introduction of position encoding reduced the requirement of time series as input. Using time and spectra as inputs to the model ensured the correct correspondence between image acquisition time and crop spectra in the time series, and also enriched the inputs to the model.</p>
<p>The unlabeled remote sensing data were very easy to obtain, and only some areas were randomly selected for pre-training in this paper, without considering the influence of the land cover degree of the area where the unlabeled remote sensing data were located on the results. Therefore, it was the focus of future work to fully exploit the potential of unlabeled remote sensing data in crop classification, including the effects of different types of unlabeled remote sensing data and remote sensing data of different time series length on the classification results. Data augmentation was an important tool for enriching sample types and avoiding model overfitting (<xref ref-type="bibr" rid="B56">Vulli et&#xa0;al., 2022</xref>), which would also be applied to crop classification in future work. In addition, the experimental results across regions were not satisfactory, and how to solve the effective migration of the model in large scale crop classification was also worthy of attention.</p>
</sec>
<sec id="s7" sec-type="conclusions">
<label>7</label>
<title>Conclusion</title>
<p>To build deep learning models that can be applied to multi-scene crop classification, we created a two-step classification system and proposed a new deep learning architecture, Cropformer. Cropformer can adapt irregular time series as input and can accumulate crop growth information in the pre-training phase, which enabled it to achieve the best performance in multiple crop classification scenarios. In full-season crop classification experiments, the average OA/AA of Cropformer with Transformer and convolution (83.50%/70.19%) outperformed traditional classification methods RF (80.73%/58.36%), the single convolution structure Res-18 (78.88%/43.38%) and the single Transformer structure of SIFT-BERT (81.70%/69.50%), indicating that Cropformer had the ability to extract more comprehensive features by using both the Convolution structure to extract local features and the Transformer to capture global information. The results of in-season crop classification experiments showed that Cropformer can obtain classification results comparable to those of crop maturity (end of September) at mid-to-late crop growth (end of July), reflecting Cropformer&#x2019;s ability of early identification, taking advantage of the ability to use irregular time series directly, and thus extracting usable features from limited images. In the classification of crops with few samples, the average OA of only 1% of the samples used by Cropformer for each class reached 82.28%, which was 2.37% lower than that of all the samples (84.65%). This showed that after pre-training, Cropformer had accumulated a lot of prior knowledge, which can effectively reduce the demand for standard label samples, so that it can obtain high-precision classification results with few labeled samples. The results of spatial transfer experiments showed that Cropformer can overcome the problem of inconsistency between the regions of the pre-training data and the fine-tuning data, indicating that Cropformer had the ability of spatial generalization. Crop mapping results showed that Cropformer can obtain mapping results consistent with field samples, which benefited from Cropformer&#x2019;s strong learning ability and can learn generalized features. All experiments showed that Cropformer adapts to multi-scenario crop classification and had great potential in large-scale crop classification.</p>
</sec>
<sec id="s8" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s9" sec-type="author-contributions">
<title>Author contributions</title>
<p>Th HW, WC, YY, ZY, YZ, SL, ZL, and XZ conducted the field experiment. HW conducted the image analysis. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec id="s10" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported in part by the National Natural Youth Science Foundation of China under Grant 42001352; in part by the National Key Research and Development Program of China under Grant 2021YFE0205100.</p>
</sec>
<sec id="s11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2023.1130659/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2023.1130659/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list id="ref1">
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdullah</surname> <given-names>A. Y. M.</given-names>
</name>
<name>
<surname>Masrur</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Adnan</surname> <given-names>M. S. G.</given-names>
</name>
<name>
<surname>Baky</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hassan</surname> <given-names>Q. K.</given-names>
</name>
<name>
<surname>Dewan</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Spatio-temporal patterns of land use/land cover change in the heterogeneous coastal region of Bangladesh between 1990 and 2017</article-title>. <source>Remote Sens.</source> <volume>11</volume>, <elocation-id>790</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs11070790</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akbar</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ullah</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>R. U.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>An effective deep learning approach for the classification of bacteriosis in peach leave</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.1064854</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ba</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Kiros</surname> <given-names>J. R.</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>G. E.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Layer normalization</article-title>. <source>arXiv preprint arXiv:1607.06450</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1607.06450</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Deep learning-based classification of hyperspectral data</article-title>. <source>IEEE J. Selected topics Appl. Earth observations Remote Sens.</source> <volume>7</volume>, <fpage>2094</fpage>&#x2013;<lpage>2107</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JSTARS.2014.2329330</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Suen</surname> <given-names>H. P.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Stable classification with limited sample: Transferring a 30-m resolution sample set collected in 2015 to mapping 10-m resolution global land cover in 2017</article-title>. <source>Sci. Bull.</source> <volume>64</volume>, <fpage>370</fpage>&#x2013;<lpage>373</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scib.2019.03.002</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choromanski</surname> <given-names>K. M.</given-names>
</name>
<name>
<surname>Likhosherstov</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Dohan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gane</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sarlos</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Rethinking attention with performers</article-title>,&#x201d; in <source>International conference on learning representations</source>. <fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2009.14794</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Constantin</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Fauvel</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Girard</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Joint supervised classification and reconstruction of irregularly sampled satellite image times series</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2021.3076667</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dauphin</surname> <given-names>Y. N.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Auli</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Grangier</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Language modeling with gated convolutional networks</article-title>,&#x201d; in <source>International conference on machine learning: PMLR.</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>Cambridge MA: JMLR</publisher-name>), <fpage>933</fpage>&#x2013;<lpage>941</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1612.08083</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>M.-W.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Bert: Pre-training of deep bidirectional transformers for language understanding</article-title>. <source>arXiv preprint arXiv:1810.04805</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1810.04805</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dosovitskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Beyer</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Kolesnikov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Weissenborn</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Unterthiner</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>An image is worth 16x16 words: Transformers for image recognition at scale</article-title>. <source>arXiv preprint arXiv:2010.11929</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eudes Gbodjo</surname> <given-names>Y. J.</given-names>
</name>
<name>
<surname>Ienco</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Leroux</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Toward spatio-spectral analysis of sentinel-2 time series data for land cover mapping</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>17</volume>, <fpage>307</fpage>&#x2013;<lpage>311</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Lgrs.2019.2917788</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname> <given-names>S. W.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>T. T.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H. Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z. X.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>X. Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Crop type identification and mapping using machine learning algorithms and sentinel-2 time series data</article-title>. <source>IEEE J. Selected Topics Appl. Earth Observations Remote Sens.</source> <volume>12</volume>, <fpage>3295</fpage>&#x2013;<lpage>3306</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Jstars.2019.2922469</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gulati</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chiu</surname> <given-names>C.-C.</given-names>
</name>
<name>
<surname>Parmar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Conformer: Convolution-augmented transformer for speech recognition</article-title>. <source>arXiv preprint arXiv:2005.08100</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2005.08100</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hao</surname> <given-names>P. Y.</given-names>
</name>
<name>
<surname>Di</surname> <given-names>L. P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>L. Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Transfer learning for crop classification with cropland data layer data (CDL) as training samples</article-title>. <source>Sci. Total Environ.</source> <volume>733</volume>, <elocation-id>138869</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scitotenv.2020.138869</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hao</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Early-season crop mapping using improved artificial immune network (IAIN) and sentinel data</article-title>. <source>PeerJ</source> <volume>6</volume>, <elocation-id>e5431</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.7717/peerj.5431</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition.</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>770</fpage>&#x2013;<lpage>778</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hua</surname> <given-names>B.-S.</given-names>
</name>
<name>
<surname>Tran</surname> <given-names>M.-K.</given-names>
</name>
<name>
<surname>Yeung</surname> <given-names>S.-K.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Pointwise convolutional neural networks</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition.</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>984</fpage>&#x2013;<lpage>993</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2018.00109</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Urban land-use mapping using a deep convolutional neural network with high spatial resolution multispectral remote sensing imagery</article-title>. <source>Remote Sens. Environ.</source> <volume>214</volume>, <fpage>73</fpage>&#x2013;<lpage>86</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2018.04.050</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ienco</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Gaetano</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Dupaquier</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Maurel</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Land cover classification <italic>via</italic> multitemporal spatial data by deep recurrent neural networks</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>14</volume>, <fpage>1685</fpage>&#x2013;<lpage>1689</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Lgrs.2017.2728698</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ioffe</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Batch normalization: Accelerating deep network training by reducing internal covariate shift</article-title>,&#x201d; in <source>International conference on machine learning: PMLR.</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>Cambridge MA: JMLR</publisher-name>), <fpage>448</fpage>&#x2013;<lpage>456</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>3D convolutional neural networks for crop classification with multi-temporal remote sensing images</article-title>. <source>Remote Sens.</source> <volume>10</volume>, <elocation-id>75</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs10010075</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khaki</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Archontoulis</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A cnn-rnn framework for crop yield prediction</article-title>. <source>Front. Plant Sci.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2019.01750</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khatami</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Mountrakis</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Stehman</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A meta-analysis of remote sensing research on supervised pixel-based land-cover image classification processes: General guidelines for practitioners and future research</article-title>. <source>Remote Sens. Environ.</source> <volume>177</volume>, <fpage>89</fpage>&#x2013;<lpage>100</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2016.02.028</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kussul</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Lavreniuk</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Skakun</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Shelestov</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Deep learning classification of land cover and crop types using remote sensing data</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>14</volume>, <fpage>778</fpage>&#x2013;<lpage>782</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/LGRS.2017.2681128</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lan</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Goodman</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gimpel</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Soricut</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Albert: A lite bert for self-supervised learning of language representations</article-title>. <source>arXiv preprint arXiv:1909.11942</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1909.11942</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lebourgeois</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Dupuy</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Vintrou</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Ameline</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Butler</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Begue</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A combined random forest and OBIA classification scheme for mapping smallholder agriculture at different nomenclature levels using multisource data (Simulated sentinel-2 time series, VHRS and DEM)</article-title>. <source>Remote Sens.</source> <volume>9</volume> (<issue>3</issue>), <elocation-id>259</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs9030259</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Z. T.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>G. K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>T. X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A CNN-transformer hybrid approach for crop classification using multitemporal multisensor images</article-title>. <source>IEEE J. Selected Topics Appl. Earth Observations Remote Sens.</source> <volume>13</volume>, <fpage>847</fpage>&#x2013;<lpage>858</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Jstars.2020.2971763</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Feichtenhofer</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Darrell</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>A ConvNet for the 2020s</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>. (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>11976</fpage>&#x2013;<lpage>11986</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01167</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Mapping cropping intensity in China using time series landsat and sentinel-2 images and Google earth engine</article-title>. <source>Remote Sens. Environ.</source> <volume>239</volume>, <elocation-id>111624</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2019.111624</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xi</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Cross-year reuse of historical samples for crop mapping based on environmental similarity</article-title>. <source>Front. Plant Sci.</source> <volume>12</volume>, <elocation-id>3447</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2021.761148</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Low</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Michel</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Dech</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Conrad</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Impact of feature selection on the accuracy and spatial uncertainty of per-field crop classification using support vector machines</article-title>. <source>Isprs J. Photogrammetry Remote Sens.</source> <volume>85</volume>, <fpage>102</fpage>&#x2013;<lpage>119</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.isprsjprs.2013.08.007</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marcos</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Volpi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kellenberger</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Tuia</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Land cover mapping at very high resolution with rotation equivariant CNNs: Towards small yet accurate models</article-title>. <source>Isprs J. Photogrammetry Remote Sens.</source> <volume>145</volume>, <fpage>96</fpage>&#x2013;<lpage>107</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.isprsjprs.2018.01.021</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martinez</surname> <given-names>J.</given-names>
</name>
<name>
<surname>La Rosa</surname> <given-names>L. E. C.</given-names>
</name>
<name>
<surname>Feitosa</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sanches</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Happ</surname> <given-names>P. N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Fully convolutional recurrent networks for multidate crop recognition from multitemporal image sequences</article-title>. <source>ISPRS J. Photogrammetry Remote Sens.</source> <volume>171</volume>, <fpage>188</fpage>&#x2013;<lpage>201</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.isprsjprs.2020.11.007</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Minh</surname> <given-names>D. H. T.</given-names>
</name>
<name>
<surname>Ienco</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Gaetano</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Lalande</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Ndikumana</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Osman</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Deep recurrent neural networks for winter vegetation quality mapping <italic>via</italic> multitemporal SAR sentinel-1</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>15</volume>, <fpage>464</fpage>&#x2013;<lpage>468</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/LGRS.2018.2794581</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mou</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Bruzzone</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X. X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Learning spectral-spatial-temporal features <italic>via</italic> a recurrent convolutional neural network for change detection in multispectral imagery</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>57</volume>, <fpage>924</fpage>&#x2013;<lpage>935</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2018.2863224</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mou</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X. X.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>A recurrent convolutional neural network for land cover change detection in multispectral images</article-title>,&#x201d; in <source>IGARSS 2018-2018 IEEE international geoscience and remote sensing symposium</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4363</fpage>&#x2013;<lpage>4366</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nevavuori</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Narra</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Lipping</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Crop yield prediction with deep convolutional neural networks</article-title>. <source>Comput. Electron. Agric.</source> <volume>163</volume>, <elocation-id>104859</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2019.104859</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Papadomanolaki</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Verma</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Vakalopoulou</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Karantzalos</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Detecting urban changes with recurrent neural networks from multitemporal sentinel-2 data</article-title>,&#x201d; in <source>IGARSS 2019-2019 IEEE international geoscience and remote sensing symposium.</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>214</fpage>&#x2013;<lpage>217</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/IGARSS.2019.8900330</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pelletier</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Valero</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Inglada</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Champion</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Dedieu</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Assessing the robustness of random forests to map land cover with high resolution satellite image time series over large areas</article-title>. <source>Remote Sens. Environ.</source> <volume>187</volume>, <fpage>156</fpage>&#x2013;<lpage>168</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2016.10.010</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Petitjean</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Inglada</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Gan&#xe7;arski</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Satellite image time series analysis under time warping</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>50</volume>, <fpage>3081</fpage>&#x2013;<lpage>3095</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2011.2179050</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rajendran</surname> <given-names>G. B.</given-names>
</name>
<name>
<surname>Kumarasamy</surname> <given-names>U. M.</given-names>
</name>
<name>
<surname>Zarro</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Divakarachari</surname> <given-names>P. B.</given-names>
</name>
<name>
<surname>Ullo</surname> <given-names>S. L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Land-use and land-cover classification using a human group-based particle swarm optimization algorithm with an LSTM classifier on hybrid pre-processing remote-sensing images</article-title>. <source>Remote Sens.</source> <volume>12</volume>, <elocation-id>4135</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs12244135</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramachandran</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zoph</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Le</surname> <given-names>Q. V.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Searching for activation functions</article-title>. <source>arXiv preprint arXiv:1710.05941</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1710.05941</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ru&#xdf;wurm</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Korner</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Temporal vegetation modelling using long short-term memory networks for crop identification from medium-resolution multi-spectral satellite images</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition workshops.</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>11</fpage>&#x2013;<lpage>19</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPRW.2017.193</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sakamoto</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Van Nguyen</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Ohno</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Ishitsuka</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Yokozawa</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Spatio&#x2013;temporal distribution of rice phenology and cropping systems in the Mekong delta with special reference to the seasonal water flow of the Mekong and bassac rivers</article-title>. <source>Remote Sens. Environ.</source> <volume>100</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2005.09.007</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Land cover classification from multi-temporal, multi-spectral remotely sensed imagery using patch-based recurrent neural networks</article-title>. <source>Neural Networks</source> <volume>105</volume>, <fpage>346</fpage>&#x2013;<lpage>355</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neunet.2018.05.019</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>An assessment of algorithmic parameters affecting image classification accuracy by random forests</article-title>. <source>Photogrammetric Eng. Remote Sens.</source> <volume>82</volume>, <fpage>407</fpage>&#x2013;<lpage>417</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.14358/Pers.82.6.407</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoaib</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ullah</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>a). <article-title>Deep learning-based segmentation and classification of leaf images for detection of tomato plant disease</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.1031748</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoaib</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ullah</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Alenezi</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>b). <article-title>A deep learning-based model for plant lesion segmentation, subtype identification, and survival probability estimation</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.1095547</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simonneaux</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Duchemin</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Helson</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Er-Raki</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Olioso</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chehbouni</surname> <given-names>A. G.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>The use of high-resolution image time series for crop classification and evapotranspiration estimate over an irrigated area in central Morocco</article-title>. <source>Int. J. Remote Sens.</source> <volume>29</volume>, <fpage>95</fpage>&#x2013;<lpage>116</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/01431160701250390</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soudani</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Le Maire</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Dufrene</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Francois</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Delpierre</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Ulrich</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2008</year>). <article-title>Evaluation of the onset of green-up in temperate deciduous broadleaf forests derived from moderate resolution imaging spectroradiometer (MODIS) data</article-title>. <source>Remote Sens. Environ.</source> <volume>112</volume>, <fpage>2643</fpage>&#x2013;<lpage>2655</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2007.12.004</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srivastava</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Krizhevsky</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Salakhutdinov</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Dropout: a simple way to prevent neural networks from overfitting</article-title>. <source>J. Mach. Learn. Res.</source> <volume>15</volume>, <fpage>1929</fpage>&#x2013;<lpage>1958</lpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
</person-group> &#x201c;<article-title>Image super-resolution via deep recursive residual network</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition.</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3147</fpage>&#x2013;<lpage>3155</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2017.298</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarasiou</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zafeiriou</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Embedding earth: Self-supervised contrastive pre-training for dense land cover classification</article-title>. <source>arXiv preprint arXiv:2203.06041</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2203.06041</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thenmozhi</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Reddy</surname> <given-names>U. S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Crop pest classification based on deep convolutional neural network and transfer learning</article-title>. <source>Comput. Electron. Agric.</source> <volume>164</volume>, <elocation-id>104906</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2019.104906</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname> <given-names>A. N.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vulli</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Srinivasu</surname> <given-names>P. N.</given-names>
</name>
<name>
<surname>Sashank</surname> <given-names>M. S. K.</given-names>
</name>
<name>
<surname>Shafi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ijaz</surname> <given-names>M. F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Fine-tuned DenseNet-169 for breast cancer metastasis prediction using FastAI and 1-cycle policy</article-title>. <source>Sensors</source> <volume>22</volume>, <elocation-id>2988</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s22082988</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>CC-SSL: A self-supervised learning framework for crop classification with few labeled samples</article-title>. <source>IEEE J. Selected Topics Appl. Earth Observations Remote Sens.</source> <volume>15</volume>, <fpage>8704</fpage>&#x2013;<lpage>8718</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JSTARS.2022.3211994</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Steiner</surname> <given-names>J. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Mapping sugarcane plantation dynamics in guangxi, China, by time series sentinel-1, sentinel-2 and landsat images</article-title>. <source>Remote Sens. Environ.</source> <volume>247</volume>, <elocation-id>111951</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2020.111951</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>J. F.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>R. H.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z. X.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>DeepCropMapping: A multi-temporal deep learning approach with improved spatial generalizability for dynamic corn and soybean mapping</article-title>. <source>Remote Sens. Environ.</source> <volume>247</volume>, <elocation-id>111946</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2020.111946</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yi</surname> <given-names>Z. W.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Q. T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Crop classification using multi-temporal sentinel-2 data in the shiyang river basin of China</article-title>. <source>Remote Sens.</source> <volume>12</volume> (<issue>24</issue>), <elocation-id>4052</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs12244052</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Self-supervised pretraining of transformers for satellite image time series classification</article-title>. <source>IEEE J. Selected Topics Appl. Earth Observations Remote Sens.</source> <volume>14</volume>, <fpage>474</fpage>&#x2013;<lpage>487</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Jstars.2020.3036602</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Q. S.</given-names>
</name>
<name>
<surname>Hang</surname> <given-names>R. L.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Z. G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>SITS-former: A pre-trained spatio-spectral-temporal representation model for sentinel-2 time series classification</article-title>. <source>Int. J. Appl. Earth Observation Geoinformation</source> <volume>106</volume>, <elocation-id>102651</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jag.2021.102651</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wardlow</surname> <given-names>B. D.</given-names>
</name>
<name>
<surname>Xiang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A review of vegetation phenological metrics extraction using time-series, multispectral satellite data</article-title>. <source>Remote Sens. Environ.</source> <volume>237</volume>, <elocation-id>111511</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2019.111511</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>EdgeFormer: Improving light-weight ConvNets by learning from vision transformers</article-title>. <source>arXiv preprint arXiv:2203.03952</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2203.03952</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Development of a global 30 m impervious surface map using multisource and multitemporal remote sensing datasets with the Google earth engine platform</article-title>. <source>Earth System Sci. Data</source> <volume>12</volume>, <fpage>1625</fpage>&#x2013;<lpage>1648</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5194/essd-12-1625-2020</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Di</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Phenological metrics-based crop classification using HJ-1 CCD images and landsat 8 imagery</article-title>. <source>Int. J. Digital Earth</source> <volume>11</volume>, <fpage>1219</fpage>&#x2013;<lpage>1240</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/17538947.2017.1387296</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Transfer learning with fully pretrained deep convolution networks for land-use classification</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>14</volume>, <fpage>1436</fpage>&#x2013;<lpage>1440</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/LGRS.2017.2691013</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname> <given-names>L. H.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>L. N.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep learning based multi-temporal crop classification</article-title>. <source>Remote Sens. Environ.</source> <volume>221</volume>, <fpage>430</fpage>&#x2013;<lpage>443</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2018.11.032</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>