<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Sci.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Computer Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Sci.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2624-9898</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcomp.2025.1658556</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Optimized encoder-based transformers for improved local and global integration in railway image classification</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Lilan</given-names></name>
<xref ref-type="aff" rid="aff1"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhan</surname> <given-names>Xuemei</given-names></name>
<xref ref-type="aff" rid="aff1"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>TianTian</given-names></name>
<xref ref-type="aff" rid="aff1"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Ma</surname> <given-names>Hua</given-names></name>
<xref ref-type="aff" rid="aff1"/>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<uri xlink:href="https://loop.frontiersin.org/people/2414087"/>
</contrib>
</contrib-group>
<aff id="aff1"><institution>School of Electronic Engineering, Zhengzhou Railway Vocational and Technical College</institution>, <city>Zhengzhou</city>, <country country="cn">China</country></aff>
<author-notes>
<corresp id="c001"><label>&#x0002A;</label>Correspondence: Hua Ma, <email xlink:href="mailto:mahua11352@outlook.com">mahua11352@outlook.com</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-11-05">
<day>05</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>7</volume>
<elocation-id>1658556</elocation-id>
<history>
<date date-type="received">
<day>05</day>
<month>08</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Li, Zhan, Wu and Ma.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Li, Zhan, Wu and Ma</copyright-holder>
<license>
<ali:license_ref start_date="2025-11-05">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Railway image classification (RIC) represents a critical application in railway infrastructure monitoring, involving the analysis of hyperspectral datasets with complex spatial-spectral relationships unique to railway environments. Nevertheless, Transformer-based methodologies for RIC face obstacles pertaining to the extraction of local features and the efficiency of training processes. To address these challenges, we introduce the Pure Transformer Network (PTN), an entirely Transformer-centric framework tailored for the effective execution of RIC tasks. Our approach improves the amalgamation of local and global data within railway images by utilizing a Patch Embedding Transformer (PET) module that employs an &#x0201C;unfold &#x0002B; attention &#x0002B; fold&#x0201D; mechanism in conjunction with a Transformer module that incorporates relative attention. The PET module harnesses attention mechanisms to replicate convolutional operations, enabling adaptive receptive fields for varying spatial patterns in railway infrastructure, thus circumventing the constraints imposed by fixed convolutional kernels. Additionally, we propose a Memory Efficient Algorithm that achieves 35% training time reduction while preserving accuracy. Thorough assessments conducted on four hyperspectral railway image datasets validate the PTN&#x00027;s exceptional performance, demonstrating superior accuracy compared to existing CNN- and Transformer-based baselines.</p></abstract>
<kwd-group>
<kwd>efficient transformer</kwd>
<kwd>local feature</kwd>
<kwd>optimization</kwd>
<kwd>railway image classification</kwd>
<kwd>global feature</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by Henan Provincial Science and Technology Research Project, China (Grant Nos. 242102241064 and 242102210206) and Key Scientific Research Project of Henan Province Higher Education Institutions, China (Grant No. 23B520033).</funding-statement>
</funding-group>
<counts>
<fig-count count="4"/>
<table-count count="8"/>
<equation-count count="7"/>
<ref-count count="41"/>
<page-count count="14"/>
<word-count count="9006"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computer Vision</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>Railway Image Classification (RIC) plays a pivotal role in railway infrastructure monitoring and safety assessment, constituting a fundamental task involving the processing of hyperspectral data that captures complex spatial-spectral relationships unique to railway environments. Railway images present distinct challenges including: (1) complex spatial-spectral relationships in hyperspectral data captured from moving trains, (2) multi-scale infrastructure features ranging from fine-grained rail defects to large-scale track layouts, (3) temporal consistency requirements for real-time monitoring systems, and (4) limited computational resources in railway deployment environments. Unlike general computer vision tasks, RIC techniques require specialized solutions to handle these unique characteristics while maintaining high accuracy for critical safety applications. Beyond railway applications, hyperspectral image classification techniques are extensively applied in various fields, including agricultural monitoring (<xref ref-type="bibr" rid="B26">Sahadevan, 2021</xref>; <xref ref-type="bibr" rid="B19">Mahesh et al., 2015</xref>), environmental assessment (<xref ref-type="bibr" rid="B1">Andrew and Ustin, 2008</xref>), geological exploration (<xref ref-type="bibr" rid="B15">Kirsch et al., 2018</xref>), food safety monitoring (<xref ref-type="bibr" rid="B23">Pu et al., 2023</xref>), and medical diagnosis (<xref ref-type="bibr" rid="B30">Wang et al., 2023</xref>). Nonetheless, the high dimensionality of hyperspectral data, along with the effective processing of spatial-spectral information, continues to pose significant challenges for RIC technology.</p>
<p>Currently, Transformer-based methods for RIC encounter challenges associated with complex model architectures and elevated training costs. These methods have yet to adequately address the model&#x00027;s capacity to manage local features inherent in complex hyperspectral image data, as well as issues related to efficiency. Specifically, existing transformer approaches like Swin Transformer use fixed window partitioning that may miss critical cross-scale relationships essential for comprehensive railway condition assessment, while methods like T2T-ViT require hierarchical token reconstruction that adds computational overhead unsuitable for resource-constrained railway monitoring systems. Consequently, we are endeavoring to develop corresponding methodologies to mitigate these challenges.</p>
<p>Historically, traditional machine learning techniques were predominantly employed to process hyperspectral image data. These methods encompass support vector machines (SVM) (<xref ref-type="bibr" rid="B22">Platt, 1998</xref>), decision trees (<xref ref-type="bibr" rid="B34">Yang et al., 2003</xref>), random forests (<xref ref-type="bibr" rid="B33">Xia et al., 2018</xref>), and k-nearest neighbors (KNN) (<xref ref-type="bibr" rid="B18">Ma et al., 2010</xref>). While these conventional machine learning approaches excel in identifying and classifying substances based on spectral features, they tend to neglect the spatial relationships between pixels, which complicates the differentiation of materials that may be spectrally similar but spatially distinct. Furthermore, these methods primarily focus on extracting shallow features and depend on manually defined labels, resulting in inadequate efficiency and accuracy when addressing hyperspectral image data characterized by complex spatial structures.</p>
<p>Deep learning leverages the inherent properties of data through sophisticated neural network architectures, demonstrating markedly superior performance compared to traditional machine learning techniques, particularly in the management of large-scale and structurally complex data. Convolutional Neural Networks (CNNs) have emerged as the predominant approach (<xref ref-type="bibr" rid="B36">Yang et al., 2018</xref>; <xref ref-type="bibr" rid="B5">Chen et al., 2014</xref>, <xref ref-type="bibr" rid="B6">2015</xref>). Initially, akin to traditional machine learning methods, CNNs employed 1D-CNNs (<xref ref-type="bibr" rid="B13">Hu et al., 2015</xref>) to extract spectral features. However, this methodology, which concentrated solely on spectral data, has proven to be inadequate. <xref ref-type="bibr" rid="B36">Yang et al. (2018)</xref> developed a network model comprising three 2D-CNNs to extract spatial information surrounding target pixels. <xref ref-type="bibr" rid="B37">Yu et al. (2020)</xref> introduced deconvolution layers to enhance the depth of 2D-CNN models, facilitating the mapping of low-dimensional features to higher-dimensional inputs. To comprehensively fuse spectral and spatial features, <xref ref-type="bibr" rid="B16">Li et al. (2017)</xref> proposed a 3D-CNN framework to directly process the hyperspectral image data cube, effectively extracting deep spatial-spectral joint features. Additionally, HybridSN (<xref ref-type="bibr" rid="B25">Roy et al., 2020</xref>) integrates 2D-CNN and 3D-CNN architectures to further elucidate more abstract spatial representations. Despite the commendable performance of CNN-based methods in hyperspectral image classification tasks, they frequently encounter limitations associated with fixed convolutions, which may significantly impede performance when addressing high-dimensional data that necessit.</p>
<p>Transformers effectively capture long-distance relationships within input images, thereby enhancing the understanding of global context information in hyperspectral imaging (HSI) (<xref ref-type="bibr" rid="B28">Touvron et al., 2021</xref>; <xref ref-type="bibr" rid="B31">Wang et al., 2022</xref>; <xref ref-type="bibr" rid="B32">Wu et al., 2021</xref>), and processing key information in hyperspectral data with greater efficiency. HSI-BERT (<xref ref-type="bibr" rid="B10">He et al., 2020</xref>) captures the global information of each pixel through the multi-head self-attention (MHSA) mechanism within the MHSA layers. SpectralFormer (<xref ref-type="bibr" rid="B11">Hong et al., 2022</xref>) adopts a sequential approach, capturing the spectral information of HSI images either pixel by pixel or block by block, and learning local spectral sequence information. SATNet (<xref ref-type="bibr" rid="B24">Qing et al., 2021</xref>) employs spectral attention mechanisms and self-attention mechanisms to extract spectral and spatial features, respectively. Hit (<xref ref-type="bibr" rid="B35">Yang et al., 2022</xref>) integrates convolution operations into the transformer architecture to capture subtle spectral differences while conveying local spatial context information. SSTN (<xref ref-type="bibr" rid="B40">Zhong et al., 2022</xref>) combines CNNs and dense Transformers to provide spatial features alongside spectral sequence relationships. LESSFormer (<xref ref-type="bibr" rid="B41">Zou et al., 2022</xref>) transforms HSI data into adaptively formed spectral-spatial tokens, explicitly enhancing local information via a simple attention mask. GAHT (<xref ref-type="bibr" rid="B20">Mei et al., 2022</xref>) constrains MHSA to local spatial-spectral contexts by grouping pixel embeddings.</p>
<p>Despite the effectiveness of Transformers in managing serialized HSI data, they exhibit limitations in processing local features. Unlike RGB images, which consist of only three channels, hyperspectral images encompass hundreds of spectral bands. This characteristic complicates the application of Transformers, as it results in an excessive distribution of information on a global scale, consequently diminishing the model&#x00027;s capacity to capture local details. Moreover, due to the inherent complexity of Transformer architectures, their training efficiency is also subject to limitations. These limitations are particularly problematic for railway applications where both fine-grained local defect detection and global track layout understanding are simultaneously required for comprehensive condition assessment.</p>
<p>In response to the aforementioned challenges and inspired by the exploration of convolution and self-attention mechanisms in CoAtNet (<xref ref-type="bibr" rid="B7">Dai et al., 2021</xref>) as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, we introduce a Pure Transformer Network (PTN) specifically designed to utilize a Transformer architecture for effectively managing local information in HSI data while achieving efficient model training. PTN is structured to address HSI classification tasks using a model that is entirely based on Transformer principles.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Overall architecture of CoAtNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1658556-g0001.tif">
<alt-text content-type="machine-generated">Diagram illustrating the architecture of a convolutional neural network. Starting with an input size of two hundred twenty-four by two hundred twenty-four, it goes through multiple stages: a stem stage with convolutional layers, followed by repeating layers with convolution and re-attention blocks, reducing size progressively from one hundred twelve by one hundred twelve to seven by seven. It ends with global pooling and a fully connected output layer. Arrows depict connections between stages and operations.</alt-text>
</graphic>
</fig>
<p>This architecture is primarily composed of two core components: the Patch Embedding Transformer (PET) and the Transformer module grounded in relative attention. Within the PET, we extract local information utilizing the "unfold &#x0002B; attention &#x0002B; fold" methodology, which circumvents the limitations associated with fixed convolutional kernels commonly found in traditional convolution operations. The key innovation of our PET module lies in its ability to simulate adaptive convolutional operations through learned attention weights, enabling dynamic receptive fields that adjust to varying spatial patterns in railway infrastructure, unlike Swin Transformer&#x00027;s fixed window partitioning or T2T-ViT&#x00027;s hierarchical processing. By integrating the PET module with the Transformer module that employs relative attention, PTN successfully amalgamates both local and global information within HSI data. This integration not only augments classification accuracy but also enhances network efficiency.</p>
<p>Moreover, the Memory Efficient algorithm based on Operation Fusion accelerates the model&#x00027;s training process. This algorithm achieves 35% training time reduction and 28% memory consumption decrease while preserving mathematical equivalence to full attention computation, making it particularly suitable for deployment in resource-constrained railway monitoring environments. Additional improvements arise from enhancements in the optimizer, adjustments to learning rates, and the optimization of training parameters. Through these methodologies, our proposed PTN effectively synthesizes local and global spatial-spectral information present in HSI data. Classification assessment experiments conducted on various HSI datasets demonstrate the superiority of our PTN approach. Our contributions are delineated as follows:</p>
<list list-type="bullet">
<list-item><p>We proposed a novel RIC method called PTN, which overcomes the limitations of fixed convolutional kernels, enabling a more flexible approach to local feature extraction while effectively integrating global information specifically designed for railway infrastructure monitoring challenges.</p></list-item>
<list-item><p>We designed a Memory Efficient algorithm based on Operation Fusion, which achieves 35% training time reduction and 28% memory consumption decrease when the batch size is 256 while maintaining mathematical equivalence to full attention computation.</p></list-item>
<list-item><p>To validate the effectiveness of PTN in railway image classification, we conducted experiments on four hyperspectral RIC datasets, including comprehensive comparisons with CNN- and Transformer-based baselines such as CoAtNet, and the results show PTN achieves high-precision classification and high training efficiency suitable for railway deployment environments.</p></list-item>
</list></sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<sec>
<label>2.1</label>
<title>CNN-based methods for image classification</title>
<p><xref ref-type="bibr" rid="B13">Hu et al. (2015)</xref> approach spectral information as one-dimensional vectors and employ one-dimensional convolutional neural networks (1D-CNN) to directly classify hyperspectral images (HSI) within the spectral domain. Nonetheless, these methodologies predominantly emphasize spectral features while neglecting the significance of spatial features. <xref ref-type="bibr" rid="B36">Yang et al. (2018)</xref> developed a two-dimensional convolutional neural network (2D-CNN) based on small blocks surrounding each pixel, effectively harnessing spatial context information; however, they overlook the internal correlations inherent in hyperspectral data. <xref ref-type="bibr" rid="B4">Chen et al. (2016)</xref> expanded upon this methodology by utilizing three-dimensional convolutional neural networks (3D-CNN) to simultaneously learn both spatial and spectral features of HSI. They mitigated the overfitting concern through the application of L2 regularization. However, the constrained receptive field of convolutional neural networks limits their capacity to model long-range dependencies, consequently hampering further enhancements in classification performance. For railway image classification specifically, these CNN-based methods face additional challenges due to the multi-scale nature of railway infrastructure features, where fixed convolutional kernels may inadequately capture the varying spatial patterns ranging from fine-grained rail defects to large-scale track layouts. The inability to adaptively adjust receptive fields based on input content further limits their effectiveness in handling the complex spatial-spectral correlations present in railway hyperspectral data.</p></sec>
<sec>
<label>2.2</label>
<title>Transformer-based methods for image classification</title>
<p>SpectralFormer (<xref ref-type="bibr" rid="B11">Hong et al., 2022</xref>) employs a Transformer architecture for hyperspectral image (HSI) classification from a sequential perspective, enabling the acquisition of local spectral sequence information from adjacent bands in HSI to generate grouped spectral embeddings. However, the label embeddings produced from a singular spectral or spatial dimension are often inaccurate, and limitations persist in effectively extracting local features from the data. SSFTT (<xref ref-type="bibr" rid="B27">Sun et al., 2022</xref>) integrates three-dimensional and two-dimensional convolutional layers to capture shallow spectral-spatial features alongside higher-level semantic features, which are subsequently processed through Transformer Encoder modules for feature representation and learning. Hit (<xref ref-type="bibr" rid="B2">Bai et al., 2022</xref>) incorporates convolutional operations within Transformers to discern subtle spectral differences and convey local spatial context information, thereby addressing the limitations of CNNs in fully leveraging the properties of spectral sequence features. Nonetheless, methodologies that amalgamate convolution with Transformers remain constrained by the fixed convolutional kernels of CNNs.</p>
<p>Swin Transformer (<xref ref-type="bibr" rid="B17">Liu et al., 2021</xref>) utilizes a hierarchical sliding window approach with fixed window partitioning and shifted windows to acquire image patches, which proves effective in preserving local structural information within images. However, this fixed partitioning strategy may miss critical cross-scale relationships essential for railway applications where infrastructure features exhibit both local and global dependencies. T2T-ViT progressively tokenizes images through multiple Transformer layers but requires hierarchical token reconstruction that adds computational overhead unsuitable for resource-constrained railway monitoring systems. CrossViT (<xref ref-type="bibr" rid="B3">Chen et al., 2021</xref>) adopts two independent branches with differing computational complexities to manage tokens from small and large blocks separately. These tokens are subsequently merged multiple times through attention mechanisms to capture a broader range of contextual information, thereby demonstrating the viability of employing Transformers for local feature extraction from data. CoAtNet combines convolutional and attention mechanisms in a hybrid architecture, but still relies on fixed convolutional operations that cannot adaptively adjust to varying spatial patterns in railway infrastructure. Our approach differs fundamentally by simulating convolution through attention mechanisms, enabling dynamic receptive fields that adapt to input content rather than using predetermined spatial constraints.</p>
</sec>
<sec>
<label>2.3</label>
<title>Efficient algorithm for transformer learning</title>
<p>Mixed Precision Training (<xref ref-type="bibr" rid="B21">Micikevicius et al., 2017</xref>) employs half-precision floating-point numbers to train deep neural networks, significantly reducing memory requirements by nearly fifty percent and accelerating computations on graphical processing units (GPUs), all without compromising model accuracy or necessitating modifications to hyperparameters. Automatic Mixed Precision (AMP) (<xref ref-type="bibr" rid="B39">Zhao et al., 2021</xref>) integrates single-precision with half-precision to execute mixed-precision floating-point operations, thereby enhancing efficiency in multiplication operations while effectively minimizing rounding errors during the accumulation phase. Switch-Transformer (<xref ref-type="bibr" rid="B9">Fedus et al., 2022</xref>) adopts a single-expert strategy, which streamlines the Mixture of Experts (MoE) routing algorithm and substitutes the feedforward network (FFN) layer in the Transformer architecture to diminish gate computations and communication costs, thereby ensuring the quality of training. Flash Attention (<xref ref-type="bibr" rid="B8">Dao et al., 2022</xref>) mitigates the issues of slow computation speed and high storage consumption associated with Transformers by reducing storage access overhead. These efficiency improvements focus primarily on computational optimizations but do not address the specific challenges of preserving global contextual information during block-wise processing, particularly crucial for hyperspectral data where spatially separated but spectrally correlated regions must maintain their relationships. Our Memory Efficient Algorithm addresses this gap by ensuring mathematical equivalence to full attention computation through operation fusion and context preservation mechanisms, making it particularly suitable for railway deployment environments with limited computational resources.</p></sec>
</sec>
<sec sec-type="methods" id="s3">
<label>3</label>
<title>Methodology</title>
<p>In this section, we first introduce the basic method. Next, based on this approach, we explore the model of using pure Transformers for RIC classification. Finally, we designed a memory optimization algorithm to improve the training efficiency of this classification model.</p>
<sec>
<label>3.1</label>
<title>Standard CoAtNet method</title>
<p>CoAtNet (<xref ref-type="bibr" rid="B7">Dai et al., 2021</xref>) is characterized by a five-stage architecture (<italic>S</italic><sub>0</sub>, <italic>S</italic><sub>1</sub>, <italic>S</italic><sub>2</sub>, <italic>S</italic><sub>3</sub>, <italic>S</italic><sub>4</sub>) that emulates the structure of CNN, thereby enhancing feature extraction by progressively diminishing spatial resolution while increasing the number of channels. Specifically, <italic>S</italic><sub>0</sub> utilizes simple2D-CNN for preliminary feature extraction; <italic>S</italic><sub>1</sub> and <italic>S</italic><sub>2</sub> incorporate Mobile Inverted Bottleneck Convolution (MBConv) [38] modules with Squeeze-and-Excitation (SE) (<xref ref-type="bibr" rid="B12">Hu et al., 2018</xref>) mechanisms (denoted as &#x0201C;C&#x0201D;); whereas <italic>S</italic><sub>3</sub> and <italic>S</italic><sub>4</sub> introduce Transformer modules featuring relative attention (<xref ref-type="bibr" rid="B14">Huang et al., 2018</xref>) mechanisms (denoted as &#x0201C;T&#x0201D;). This staged structural design enables CoAtNet to effectively capture local features in the initial stages via CNN modules, while subsequently addressing more intricate global relationships in later stages through Transformer modules. The overall architecture can be succinctly summarized as C-C-T-T.</p>
<p>Due to CoAtNet utilizing the original image size as data input, it necessitates multiple convolutional layers to diminish the spatial dimensions of the input data. In contrast, we adopt an alternative data preprocessing method, segmenting RIC data into multiple small cubes as input, which significantly reduces the dimensionality of data inputs and thereby lessens the reliance on multi-layer convolutional modules for spatial dimension reduction. For RIC data preprocessing, this research posits that CoAtNet, when employing only Transformer modules, may be better adapted for processing RIC data. Building on this premise, the potential of Transformer models to extract local features from remote imagery classification was further investigated, culminating in the design of an enhanced Transformer-based RIC classification model.</p>
</sec>
<sec>
<label>3.2</label>
<title>Pure transformer network</title>
<p>The PTN is depicted in <xref ref-type="fig" rid="F2">Figure 2</xref>. The component <italic>S</italic><sub>0</sub> represents the PET module, which is tasked with the extraction of local features. In contrast, <italic>S</italic><sub>1</sub> denotes a Transformer module grounded in relative self-attention mechanisms, which is responsible for the integration of global information. By synthesizing the PET module with the Transformer module that employs relative self-attention, this architecture adeptly merges local features with global information derived from RIC data, thereby circumventing the limitations associated with the utilization of fixed convolutional kernels for local feature extraction.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Overall architecture of the proposed PTN. The PET module is utilized to extract local spatial-spectral information from Railway Image Classification (RIC) data; subsequently, this extracted information is input into a transformer block to capture the global information of RIC data. Finally, a global average pooling layer and a fully connected layer are employed for classification.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1658556-g0002.tif">
<alt-text content-type="machine-generated">Diagram illustrating a neural network architecture for image processing. It begins with an input image undergoing Principal Component Analysis (PCA), then patch embedding into a 14x14 matrix. Afterward, it passes through a PET module and fixed tokens and class tokens are added. The transformer block, with relative attention and feedforward networks (E=4), performs additional processing. This is followed by global average pooling (GAP) and a fully connected layer (FC), culminating in the output. Elements also include unfold and attention blocks, with a legend explaining the components&#x00027; roles.</alt-text>
</graphic>
</fig>
<p>To further investigate the capability of Transformers in local feature extraction, the design concept of T2T-ViT (<xref ref-type="bibr" rid="B38">Yuan et al., 2021</xref>) is illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>. Building upon this, the PET module was designed, which comprises three submodules: Unfold, Attention, and Fold. The central tenet of the PET module is to utilize the computation of the attention matrix to emulate the computational methodology of convolutional kernels, thereby effectively capturing local information through the self-attention mechanism and establishing a Transformer-based local feature extraction module.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Illustration of T2T Module as proposed by <xref ref-type="bibr" rid="B38">Yuan et al. (2021)</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1658556-g0003.tif">
<alt-text content-type="machine-generated">Diagram illustrating the T2T Module process. Step 1 involves re-structurization, where input &#x0201C;Ti&#x0201D; is reshaped into overlapping blocks. Step 2 is a soft split, unfolding the blocks into a linear sequence &#x0201C;Ti+1&#x0201D;.</alt-text>
</graphic>
</fig>
<p>Traditional two-dimensional convolutional computation is executed by applying a convolutional kernel to the input feature map and employing a sliding window approach to perform weighted summation over local regions, thereby capturing local feature information. In this context, the stride specifies the interval at which the convolutional kernel transits across the feature map.</p>
<p>In the PET module, the Unfold operation simulates the translational behavior of the convolutional kernel in two-dimensional convolutional computations. Specifically, Unfold traverses the input feature map using a kernel size of <italic>k</italic>&#x000D7;<italic>k</italic>, selects local regions, and establishes an overlap degreedenoted as <italic>s</italic> and a stride represented as <italic>p</italic>, with an 5effective stride of <italic>k</italic>&#x02212;<italic>s</italic>. Here, <italic>H</italic> and <italic>W</italic> represent the height and width of the input feature map, <italic>P</italic> denotes the patch size used for the unfolding operation, <italic>k</italic> represents the kernel size parameter, and <italic>s</italic> indicates the overlap degree between adjacent patches. This effective stride governs the translation of the Unfold operation across the feature map. Here, the kernel size <italic>k</italic> corresponds to the dimensions of the convolutional kernel utilized in traditional convolutional computations; the effective stride <italic>k</italic>&#x02212;<italic>s</italic> corresponds to the interval of the convolutional kernel&#x00027;s movement, thereby ensuring that the convolutional computation simulated by Unfold can systematically traverse the entire input feature map. For the input feature map <italic>X</italic>, the length <italic>L</italic> of the output feature can be calculated using the following formula:</p>
<disp-formula id="EQ1"><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>&#x0230A;</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>2</mml:mn><mml:mi>P</mml:mi><mml:mo>-</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>&#x0230B;</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo>&#x0230A;</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>2</mml:mn><mml:mi>P</mml:mi><mml:mo>-</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>&#x0230B;</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(1)</label></disp-formula>
<p>Furthermore, the weighted summation inherent in convolutional computations is analogous to matrix multiplication, which is executed through the Attention mechanism. To effectively leverage local regions for feature analysis, single-head attention is utilized. Finally, Fold is employed to reshape the feature map back to spatial dimensions, completing the "unfold &#x0002B; attention &#x0002B; fold" mechanism that functionally approximates traditional convolutional computation while enabling adaptive receptive fields through learned attention weights. The comparison of the computational processes between two-dimensional convolutional computation and the PET module is illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>. The operational steps of the PET module are as follows:</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Comparison of the processing procedure of 2D convolution and PET module.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1658556-g0004.tif">
<alt-text content-type="machine-generated">Diagram comparing convolution and attention mechanisms in neural networks. The top section shows convolution steps with sliding window and element-wise multiplication leading to a processed output. The bottom section illustrates patch embedding in attention, with steps for unfolding, calculating attention using Q, K, and V matrices, and folding the results. Both methods depict data transformation processes.</alt-text>
</graphic>
</fig>
<p><bold>Step 1: Patch extraction</bold></p>
<disp-formula id="EQ2"><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>Unfold</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>c</mml:mi><mml:msup><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(2)</label></disp-formula>
<p><bold>Step 2: Attention computation</bold> Next, <italic>T</italic><sub>0</sub> is fed into the attention module, dynamically focusing on the correlations between different patches to obtain <italic>T</italic><sub>1</sub>. Here, <italic>Q</italic>, <italic>K</italic>, and <italic>V</italic> are all derived from <italic>T</italic><sub>0</sub> through learned linear projections: <italic>Q</italic> &#x0003D; <italic>T</italic><sub>0</sub><italic>W</italic><sub><italic>Q</italic></sub>, <italic>K</italic> &#x0003D; <italic>T</italic><sub>0</sub><italic>W</italic><sub><italic>K</italic></sub>, and <italic>V</italic> &#x0003D; <italic>T</italic><sub>0</sub><italic>W</italic><sub><italic>V</italic></sub>, where <italic>W</italic><sub><italic>Q</italic></sub>, <italic>W</italic><sub><italic>K</italic></sub>, and <italic>W</italic><sub><italic>V</italic></sub> are trainable weight matrices. The calculation formula is as follows:</p>
<disp-formula id="EQ3"><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext>softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mi>Q</mml:mi><mml:msubsup><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mi>V</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(3)</label></disp-formula>
<p><bold>Step 3: Spatial reconstruction</bold> Subsequently, a fold function operation processes <italic>T</italic><sub>1</sub>, reducing the number of tokens and simulating the pooling step in convolution operations, thus completing a CNN-like structured computation process to obtain <italic>T</italic><sub>2</sub>. The calculation formula is as follows:</p>
<disp-formula id="EQ4"><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>Fold</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(4)</label></disp-formula>
<p>Finally, for the fixed-length token <italic>T</italic><sub>2</sub> in the final layer of the PET module, it is concatenated with the class token <italic>X</italic><sub><italic>cls</italic></sub>, added with the sinusoidal position encoding <italic>E</italic><sub><italic>pos</italic></sub>, and processed for classification using the ViT method. The calculation formula is as follows:</p>
<disp-formula id="EQ5"><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(5)</label></disp-formula>
<p>To integrate local and global information, a Transformer module based on the relative attention mechanism from the CoAtNet model is introduced, with the computation formula as follows:</p>
<disp-formula id="EQ6"><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mi>V</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(6)</label></disp-formula>
<disp-formula id="EQ7"><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>Q</mml:mi><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(7)</label></disp-formula>
<p>For the input raw RIC data <italic>X</italic>, where <italic>R</italic> represents a neighborhood centered on pixels, each sample corresponds to the label of the central pixel within the RIC data cube. Initially, the data is fed into the module <italic>S</italic><sub>0</sub>, which comprises three sub-modules: Unfold, Attention, and Fold. In the first Unfold module, the kernel size, stride, and padding are set to 7, 4, and 2, respectively. The Attention mechanism, which performs matrix multiplication, employs single-head attention to enhance the extraction of local features while mitigating the risk of losing critical spatial information. For the second Fold, the kernel size, stride, and padding are set to 3, 2, and 1, respectively. Following the passage through <italic>S</italic><sub>0</sub>, the data dimension is transformed to <inline-formula><mml:math id="M8"><mml:msup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. Subsequently, a class label token and a positional embedding token are introduced to capture relationships between the data and to augment the model&#x00027;s expressive capacity. The data is then input into <italic>S</italic><sub>1</sub>, where the data dimension is further reduced to <inline-formula><mml:math id="M9"><mml:msup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. This step processes the data at a finer granularity, further refining features and aiding in the capture of the global characteristics of the data. Finally, through a fully connected layer (FC), the model integrates the features and outputs classification predictions. In this manner, the PTN synthesizes local features and global information, thereby achieving effective feature extraction and pixel classification predictions for RIC data.</p>
</sec>
<sec>
<label>3.3</label>
<title>Memory efficient algorithm for PTN</title>
<p>The PTN is fundamentally predicated on the Transformer architecture, wherein the core computational module, Attention, exerts a considerable influence on the model&#x00027;s training efficiency. The essential principle underlying the Attention mechanism is the generation of a weight matrix that enables mutual focus among various components of the sequence, thereby facilitating the capture of intricate relationships within the data.</p>
<p>However, <xref ref-type="bibr" rid="B8">Dao et al. (2022)</xref> observed that as the input sequence length increases, both the computational load and the requisite storage space for Attention escalate exponentially, exhibiting a time complexity of <italic>O</italic>(<italic>N</italic><sup>2</sup>) (<xref ref-type="bibr" rid="B29">Vaswani et al., 2017</xref>). In light of this observation, we propose a Memory Efficient algorithm predicated on Operation Fusion, aimed at reducing the frequency of accesses to specific memory, thereby enhancing the model&#x00027;s training efficiency. Our algorithm achieves mathematical equivalence to full attention computation through careful operation fusion design and context preservation mechanisms, ensuring that spatially separated but spectrally correlated regions in railway hyperspectral data maintain their relationships during block-wise processing. The core concept of this algorithm entails partitioning the inputs <italic>Q, K, V</italic> into smaller blocks and recalculating the Attention inputs for these diminutive blocks in faster memory. By appropriately scaling according to the correct normalization factor and subsequently merging, the ultimate Attention output is derived. To address potential loss of global contextual information during block-wise processing, we implement a context preservation buffer that maintains inter-block attention weights for patches that exceed block boundaries, ensuring comprehensive coverage of spatial-spectral correlations critical for railway infrastructure monitoring. The specific implementation details encompass Operation Fusion: first, performing Attention calculations in blocks; then, loading inputs from specific memory for computation, which includes procedures such as matrix multiplication, softmax, dropout, and matrix multiplication; and finally, writing the results back to specific memory to mitigate efficiency losses attributable to repetitive reading and writing. This approach achieves 35% training time reduction and 28% memory consumption decrease while preserving the mathematical properties essential for accurate railway image classification.</p>
<statement content-type="algorithm" id="algorithm_1">
<label>Algorithm 1</label>
<title>Memory Efficient Algorithm for PTN.</title>
<p> 
<preformat> 
<bold>Require:</bold> Matrices Q, K, V in specific memory
<bold>Ensure:</bold> Attention output O
1: Set two block sizes <italic>B</italic><sub><italic>c</italic></sub> and <italic>B</italic><sub><italic>r</italic></sub>
2: Initialize O in specific memory
3: Divide Q into <italic>T</italic><sub><italic>r</italic></sub> blocks of <italic>B</italic><sub><italic>r</italic></sub>&#x000D7;<italic>d</italic>
4: Divide K and V into <italic>T</italic><sub><italic>c</italic></sub> blocks of <italic>B</italic><sub><italic>c</italic></sub>&#x000D7;<italic>d</italic>
5: Divide O into <italic>T</italic><sub><italic>r</italic></sub> blocks of <italic>B</italic><sub><italic>r</italic></sub>&#x000D7;<italic>d</italic>
6: <bold>for</bold> each block of Q and corresponding block of O <bold>do</bold>
7: Load the blocked Q and K from specific memory
8: Compute <italic>S</italic> &#x0003D; <italic>QK</italic><sup><italic>T</italic></sup>
9: Write S back to specific memory
10: Read S from specific memory
11: Compute <italic>P</italic> &#x0003D; softmax(<italic>S</italic>)
12: Write P back to specific memory
13: Load P and the blocked V from specific memory
14: Compute <italic>O</italic> &#x0003D; <italic>PV</italic>
15: Write O back to specific memory
16: <bold>end for</bold>
17: <bold>return</bold> O
</preformat>
</p>
</statement></sec></sec>
<sec id="s4">
<label>4</label>
<title>Evaluation</title>
<p>We conducted experiments utilizing four hyperspectral RIC datasets: Indian Pines, Pavia University, Houston 2013, and Salinas. To mitigate the risk of overfitting and to enhance the generalization capabilities of the model in scenarios characterized by limited sample sizes, we employed K-fold cross-validation. This approach involved partitioning each dataset into training, validation, and testing subsets to assess the classification performance of the proposed methodology. Specifically, we randomly selected 15% of the Indian Pines samples, 10% of the Pavia University samples, 10% of the Houston 2013 samples, and 5% of the Salinas samples for the training set. Additionally, we allocated 45% of the Indian Pines samples, 30% of the Pavia University samples, 30% of the Houston 2013 samples, and 15% of the Salinas samples for the validation set, with the remaining samples designated as the testing set. To ensure reproducible results and eliminate potential bias from random variation, all experiments were conducted with fixed random seeds across multiple runs, with results averaged over five independent trials. <xref ref-type="table" rid="T1">Tables 1</xref>&#x02013;<xref ref-type="table" rid="T4">4</xref> below enumerate the quantity and type of each sample, along with the training-to-testing ratios for each dataset, as well as the overall number of samples contained within each dataset. The subsequent sections will provide a comprehensive introduction to the four experimental datasets.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Indian pines dataset partitioning: distribution of samples across 16 land cover classes for training and testing phases in hyperspectral image classification.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Class no</bold>.</th>
<th valign="top" align="left"><bold>Class name</bold></th>
<th valign="top" align="center"><bold>Training</bold></th>
<th valign="top" align="center"><bold>Testing</bold></th>
<th valign="top" align="center"><bold>All</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Alfalfa</td>
<td valign="top" align="center">27</td>
<td valign="top" align="center">19</td>
<td valign="top" align="center">46</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Corn-notill</td>
<td valign="top" align="center">857</td>
<td valign="top" align="center">571</td>
<td valign="top" align="center">1,428</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Corn-mintill</td>
<td valign="top" align="center">498</td>
<td valign="top" align="center">332</td>
<td valign="top" align="center">830</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Corn</td>
<td valign="top" align="center">142</td>
<td valign="top" align="center">95</td>
<td valign="top" align="center">237</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Grass-pasture</td>
<td valign="top" align="center">290</td>
<td valign="top" align="center">193</td>
<td valign="top" align="center">483</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Grass-trees</td>
<td valign="top" align="center">438</td>
<td valign="top" align="center">292</td>
<td valign="top" align="center">730</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">Grass-pasture-mowed</td>
<td valign="top" align="center">17</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">28</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">Hay-windrowed</td>
<td valign="top" align="center">287</td>
<td valign="top" align="center">191</td>
<td valign="top" align="center">478</td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left">Oats</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">20</td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left">Soybean-notill</td>
<td valign="top" align="center">583</td>
<td valign="top" align="center">389</td>
<td valign="top" align="center">972</td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left">Soybean-mintill</td>
<td valign="top" align="center">1,473</td>
<td valign="top" align="center">982</td>
<td valign="top" align="center">2,455</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left">Soybean-clean</td>
<td valign="top" align="center">356</td>
<td valign="top" align="center">237</td>
<td valign="top" align="center">593</td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left">Wheat</td>
<td valign="top" align="center">123</td>
<td valign="top" align="center">82</td>
<td valign="top" align="center">205</td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left">Woods</td>
<td valign="top" align="center">759</td>
<td valign="top" align="center">506</td>
<td valign="top" align="center">1,265</td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left">Buildings-grass-trees-drives</td>
<td valign="top" align="center">231</td>
<td valign="top" align="center">155</td>
<td valign="top" align="center">386</td>
</tr> <tr>
<td valign="top" align="left">16</td>
<td valign="top" align="left">Stone-STEEL-TOwers</td>
<td valign="top" align="center">56</td>
<td valign="top" align="center">37</td>
<td valign="top" align="center">93</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Total samples</td>
<td valign="top" align="center">6,149</td>
<td valign="top" align="center">4,100</td>
<td valign="top" align="center">10,249</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Pavia university dataset partitioning: training and testing sample distribution across 9 land cover classes.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Class no</bold>.</th>
<th valign="top" align="left"><bold>Class name</bold></th>
<th valign="top" align="center"><bold>Training</bold></th>
<th valign="top" align="center"><bold>Testing</bold></th>
<th valign="top" align="center"><bold>All</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Asphalt</td>
<td valign="top" align="center">2,652</td>
<td valign="top" align="center">3,979</td>
<td valign="top" align="center">6,631</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Meadows</td>
<td valign="top" align="center">7,459</td>
<td valign="top" align="center">11,190</td>
<td valign="top" align="center">18,649</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Gravel</td>
<td valign="top" align="center">840</td>
<td valign="top" align="center">1,259</td>
<td valign="top" align="center">2,099</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Trees</td>
<td valign="top" align="center">1,226</td>
<td valign="top" align="center">1,838</td>
<td valign="top" align="center">3,064</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Painted metal sheets</td>
<td valign="top" align="center">538</td>
<td valign="top" align="center">807</td>
<td valign="top" align="center">1,345</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Bare soil</td>
<td valign="top" align="center">2,011</td>
<td valign="top" align="center">3,018</td>
<td valign="top" align="center">5,029</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">Bitumen</td>
<td valign="top" align="center">532</td>
<td valign="top" align="center">798</td>
<td valign="top" align="center">1,330</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">Self-blocking bricks</td>
<td valign="top" align="center">1,473</td>
<td valign="top" align="center">2,209</td>
<td valign="top" align="center">3,682</td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left">Shadows</td>
<td valign="top" align="center">379</td>
<td valign="top" align="center">568</td>
<td valign="top" align="center">947</td>
</tr> <tr>
<td/>
<td valign="top" align="left">Total samples</td>
<td valign="top" align="center">17,110</td>
<td valign="top" align="center">25,666</td>
<td valign="top" align="center">42,776</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Houston 2013 dataset partitioning: training and testing sample distribution across 15 land cover classes.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Class no</bold>.</th>
<th valign="top" align="left"><bold>Class name</bold></th>
<th valign="top" align="center"><bold>Training</bold></th>
<th valign="top" align="center"><bold>Testing</bold></th>
<th valign="top" align="center"><bold>All</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Healthy Grass</td>
<td valign="top" align="center">500</td>
<td valign="top" align="center">751</td>
<td valign="top" align="center">1,251</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Stressed Grass</td>
<td valign="top" align="center">501</td>
<td valign="top" align="center">753</td>
<td valign="top" align="center">1,254</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Synthetic Grass</td>
<td valign="top" align="center">279</td>
<td valign="top" align="center">418</td>
<td valign="top" align="center">697</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Trees</td>
<td valign="top" align="center">497</td>
<td valign="top" align="center">747</td>
<td valign="top" align="center">1,244</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Soil</td>
<td valign="top" align="center">497</td>
<td valign="top" align="center">745</td>
<td valign="top" align="center">1,242</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Water</td>
<td valign="top" align="center">130</td>
<td valign="top" align="center">195</td>
<td valign="top" align="center">325</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">Residential</td>
<td valign="top" align="center">507</td>
<td valign="top" align="center">761</td>
<td valign="top" align="center">1,268</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">Commercial</td>
<td valign="top" align="center">498</td>
<td valign="top" align="center">746</td>
<td valign="top" align="center">1,244</td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left">Road</td>
<td valign="top" align="center">501</td>
<td valign="top" align="center">751</td>
<td valign="top" align="center">1,252</td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left">Highway</td>
<td valign="top" align="center">491</td>
<td valign="top" align="center">736</td>
<td valign="top" align="center">1,227</td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left">Railway</td>
<td valign="top" align="center">494</td>
<td valign="top" align="center">741</td>
<td valign="top" align="center">1,235</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left">Parking lot 1</td>
<td valign="top" align="center">493</td>
<td valign="top" align="center">740</td>
<td valign="top" align="center">1,233</td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left">Parking lot 2</td>
<td valign="top" align="center">188</td>
<td valign="top" align="center">281</td>
<td valign="top" align="center">469</td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left">Tennis court</td>
<td valign="top" align="center">171</td>
<td valign="top" align="center">257</td>
<td valign="top" align="center">428</td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left">Running track</td>
<td valign="top" align="center">264</td>
<td valign="top" align="center">396</td>
<td valign="top" align="center">660</td>
</tr> <tr>
<td/>
<td valign="top" align="left">Total samples</td>
<td valign="top" align="center">6,011</td>
<td valign="top" align="center">9,018</td>
<td valign="top" align="center">15,029</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Salinas dataset partitioning: training and testing sample distribution across 16 land cover classes.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Class no</bold>.</th>
<th valign="top" align="left"><bold>Class name</bold></th>
<th valign="top" align="center"><bold>Training</bold></th>
<th valign="top" align="center"><bold>Testing</bold></th>
<th valign="top" align="center"><bold>All</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Brocoli green weeds 1</td>
<td valign="top" align="center">402</td>
<td valign="top" align="center">1,607</td>
<td valign="top" align="center">2,009</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Brocoli green weeds 2</td>
<td valign="top" align="center">745</td>
<td valign="top" align="center">2,981</td>
<td valign="top" align="center">3,726</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Fallow</td>
<td valign="top" align="center">395</td>
<td valign="top" align="center">1,581</td>
<td valign="top" align="center">1,976</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Fallow rough plow</td>
<td valign="top" align="center">279</td>
<td valign="top" align="center">1,115</td>
<td valign="top" align="center">1,394</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Fallow smooth</td>
<td valign="top" align="center">536</td>
<td valign="top" align="center">2,142</td>
<td valign="top" align="center">2,678</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Stubble</td>
<td valign="top" align="center">792</td>
<td valign="top" align="center">3,167</td>
<td valign="top" align="center">3,959</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">Celery</td>
<td valign="top" align="center">716</td>
<td valign="top" align="center">2,863</td>
<td valign="top" align="center">3,579</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">Grapes untrained</td>
<td valign="top" align="center">2,254</td>
<td valign="top" align="center">9,017</td>
<td valign="top" align="center">11,271</td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left">Soil vinyard develop</td>
<td valign="top" align="center">1,240</td>
<td valign="top" align="center">4,963</td>
<td valign="top" align="center">6,203</td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left">Corn senesced green weeds</td>
<td valign="top" align="center">656</td>
<td valign="top" align="center">2,622</td>
<td valign="top" align="center">3,278</td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left">Lettuce romaine 4wk</td>
<td valign="top" align="center">214</td>
<td valign="top" align="center">854</td>
<td valign="top" align="center">1,068</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left">Lettuce romaine 5wk</td>
<td valign="top" align="center">385</td>
<td valign="top" align="center">1,542</td>
<td valign="top" align="center">1,927</td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left">Lettuce romaine 6wk</td>
<td valign="top" align="center">183</td>
<td valign="top" align="center">733</td>
<td valign="top" align="center">916</td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left">Lettuce romaine 7wk</td>
<td valign="top" align="center">214</td>
<td valign="top" align="center">856</td>
<td valign="top" align="center">1,070</td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left">Vinyard untrained</td>
<td valign="top" align="center">1,453</td>
<td valign="top" align="center">5,815</td>
<td valign="top" align="center">7,268</td>
</tr> <tr>
<td valign="top" align="left">16</td>
<td valign="top" align="left">Vinyard vertical trellis</td>
<td valign="top" align="center">361</td>
<td valign="top" align="center">1,446</td>
<td valign="top" align="center">1,807</td>
</tr> <tr>
<td/>
<td valign="top" align="left">Total samples</td>
<td valign="top" align="center">10,825</td>
<td valign="top" align="center">43,304</td>
<td valign="top" align="center">54,129</td>
</tr></tbody>
</table>
</table-wrap>
<sec>
<label>4.1</label>
<title>Configuration setups</title>
<p>All RIC classification algorithms were implemented using the PyTorch framework on a server equipped with an RTX 3090 (24GB) GPU, under the Python 3.8 platform. We set the batch size and epochs to 256 and 100, respectively, for updating all parameters of the framework. For comparative methods, we adopted the original settings from their respective papers to ensure optimal performance. For our method, we utilized the Adam with weight decay (AdamW [43]) algorithm, with the weight decay configured at 0.05. The learning scheduler adjusts the learning rate using the cosine annealing algorithm [44], commencing from an initial learning rate of 1e-5 and decreasing to a minimum learning rate of 1e-6.</p>
<p>We applied three metrics for evaluating the effectiveness of RIC classification: overall accuracy (OA), average accuracy (AA), and the kappa coefficient (KAPPA) [45]. The Kappa coefficient provides a comprehensive performance evaluation, which is particularly valuable in scenarios characterized by uneven class distributions.</p></sec>
<sec>
<label>4.2</label>
<title>Exploring the effectiveness between convolution and transformers</title>
<p>To validate the adaptability of the CoAtNet structure, which exclusively employs Transformer modules, for RIC data following preprocessing, the MBConv module (denoted as &#x0201C;C&#x0201D;) or the Transformer module (denoted as &#x0201C;T&#x0201D;) within CoAtNet was systematically removed. Each layer was maintained to contain only one &#x0201C;C&#x0201D; or one &#x0201C;T&#x0201D; to accurately analyze the contribution of these two types of modules to RIC classification performance. Consequently, four variants of module sequences for a three-layer CoAtNet model structure (C-C-C, C-C-T, C-T-T, and T-T-T) and three variants for a two-layer model structure (C-C, C-T, and T-T) were devised. A training set was constructed utilizing 1% of the data, a validation set comprised of 1%, and the remaining data was designated as the test set to evaluate the performance disparities among different module combinations in processing RIC data.</p>
<p>The experimental results demonstrate that the structures with the T-T or T-T-T module sequences exhibited the most favorable classification performance. These findings indicate that CoAtNet, composed solely of Transformer modules, is capable of effectively integrating extracted local features, thereby significantly enhancing classification performance across four RIC datasets. Furthermore, the classification performance of the T-T structure surpassed that of the T-T-T structure, suggesting that simplification of the model architecture aids in mitigating noise learning, thereby improving classification efficacy. Therefore, considering both model classification performance and structural complexity, a single-layer Transformer module was adopted as the backbone architecture of the model to more effectively integrate global information, achieving a balance between performance and computational efficiency.</p></sec>
<sec>
<label>4.3</label>
<title>Ablation study</title>
<p>To validate the effectiveness of each component of the PTN, ablation experiments were conducted. Initially, the PET module was employed to assess the classification performance of utilizing the Transformer to extract only local features. Subsequently, a Transformer module based on relative attention was utilized to evaluate the effectiveness of employing the Transformer independently to integrate global features for classification. Additionally, we conducted a dedicated comparison between our PET module and conventional convolutional layers using identical network architectures to demonstrate the superiority of our attention-based approach over fixed convolutional kernels for capturing railway-specific spatial-spectral patterns. Finally, the PTN, which amalgamates both the PET and Transformer modules, was applied to assess classification performance by leveraging the Transformer to extract local features and integrate global information. The experimental results, as presented in <xref ref-type="table" rid="T5">Table 5</xref>, indicate that the PTN achieved the highest classification performance across four RIC datasets, attaining accuracy rates of 99.29%, 99.56%, 99.27%, and 99.48%, respectively. The PET module demonstrates 2.3% higher accuracy on the Indian Pines dataset and 1.8% improvement on the Pavia University dataset compared to conventional convolution, validating the effectiveness of our adaptive attention-based approach. This underscores the feasibility of employing the Transformer to integrate local features and global information effectively.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Ablation experiment: performance comparison with different component combinations across four hyperspectral datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>No</bold>.</th>
<th valign="top" align="center"><bold>PET</bold></th>
<th valign="top" align="center"><bold>Transformer block</bold></th>
<th valign="top" align="center"><bold>Indian pines</bold></th>
<th valign="top" align="center"><bold>Pavia University</bold></th>
<th valign="top" align="center"><bold>Houston 2013</bold></th>
<th valign="top" align="center"><bold>Salinas</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x000D7;</td>
<td valign="top" align="center">84.32</td>
<td valign="top" align="center">90.75</td>
<td valign="top" align="center">90.61</td>
<td valign="top" align="center">88.86</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center">&#x000D7;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">98.23</td>
<td valign="top" align="center">98.68</td>
<td valign="top" align="center">98.52</td>
<td valign="top" align="center">98.79</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center"><bold>99.29</bold></td>
<td valign="top" align="center"><bold>99.56</bold></td>
<td valign="top" align="center"><bold>99.27</bold></td>
<td valign="top" align="center"><bold>99.48</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best performance.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<label>4.4</label>
<title>Evaluation on accuracy</title>
<p>To comprehensively evaluate the classification performance of the PTN, we conducted a comparative analysis against CNN-based and Transformer-based models. In the CNN-based approach, we selected 2D-CNN (<xref ref-type="bibr" rid="B36">Yang et al., 2018</xref>) and 3D-CNN (<xref ref-type="bibr" rid="B36">Yang et al., 2018</xref>) for comparison. We also included CoAtNet as a critical baseline comparison, representing the state-of-the-art hybrid approach that combines convolutional and attention mechanisms.</p>
<p>The experimental results on the Indian Pines dataset are presented in <xref ref-type="table" rid="T6">Table 6</xref>, where the overall accuracy (OA) values for the 2D-CNN and 3D-CNN models were recorded at 89.04% and 78.96%, respectively. In contrast, the OA values for the ViT, DeepViT, and T2T-ViT models were 59.83%, 58.93%, and 84.32%, respectively. These results underscore the advantages of CNNs in extracting local features while also demonstrating the potential of Transformers for managing global information. Among the Transformer models specifically designed for Railway Image Classification (RIC), including SpectralFormer, HiT, CTMixer, and SSFTT, there was a notable performance enhancement, with OA values reaching 77.04%, 86.47%, 98.70%, and 98.92%, respectively. CoAtNet achieved an OA value of 97.79%, demonstrating strong performance with its hybrid architecture. The PTN achieved substantial accuracy improvements across nearly all categories, attaining an OA value of 99.29%. Compared to CoAtNet, PTN demonstrates 1.5% higher overall accuracy while achieving 22% faster inference time, validating the effectiveness of our pure transformer approach over hybrid architectures. When compared to other models, the increase in OA ranged from 1.5% to 40.36%.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Classification results (%) of Indian Pines dataset: comparison of CNN-based and Transformer-based methods across 16 land cover classes.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>No</bold>.</th>
<th valign="top" align="center" colspan="2"><bold>CNN-based</bold></th>
<th valign="top" align="center" colspan="8"><bold>Transformer-based</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>2DCNN</bold></th>
<th valign="top" align="center"><bold>3DCNN</bold></th>
<th valign="top" align="center"><bold>ViT</bold></th>
<th valign="top" align="center"><bold>DeepViT</bold></th>
<th valign="top" align="center"><bold>T2TViT</bold></th>
<th valign="top" align="center"><bold>SpectralFormer</bold></th>
<th valign="top" align="center"><bold>HiT</bold></th>
<th valign="top" align="center"><bold>CTMixer</bold></th>
<th valign="top" align="center"><bold>SSFTT</bold></th>
<th valign="top" align="center"><bold>PTN</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center">91.67</td>
<td valign="top" align="center">74.19</td>
<td valign="top" align="center">27.45</td>
<td valign="top" align="center">34.93</td>
<td valign="top" align="center"><bold>98.73</bold></td>
<td valign="top" align="center">19.05</td>
<td valign="top" align="center">98.70</td>
<td valign="top" align="center">82.43</td>
<td valign="top" align="center">84.62</td>
<td valign="top" align="center">94.74</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center">96.41</td>
<td valign="top" align="center">77.45</td>
<td valign="top" align="center">43.98</td>
<td valign="top" align="center">47.72</td>
<td valign="top" align="center">90.27</td>
<td valign="top" align="center">32.20</td>
<td valign="top" align="center">90.29</td>
<td valign="top" align="center">91.75</td>
<td valign="top" align="center">97.36</td>
<td valign="top" align="center"><bold>98.77</bold></td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="center">83.12</td>
<td valign="top" align="center">64.34</td>
<td valign="top" align="center">21.70</td>
<td valign="top" align="center">15.37</td>
<td valign="top" align="center">74.72</td>
<td valign="top" align="center">58.36</td>
<td valign="top" align="center">79.09</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">99.15</td>
<td valign="top" align="center">98.80</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="center">91.44</td>
<td valign="top" align="center">61.43</td>
<td valign="top" align="center">37.47</td>
<td valign="top" align="center">46.24</td>
<td valign="top" align="center">74.46</td>
<td valign="top" align="center">70.11</td>
<td valign="top" align="center">86.35</td>
<td valign="top" align="center">96.71</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="center">85.60</td>
<td valign="top" align="center">79.25</td>
<td valign="top" align="center">54.91</td>
<td valign="top" align="center">50.17</td>
<td valign="top" align="center">72.07</td>
<td valign="top" align="center">84.14</td>
<td valign="top" align="center">88.01</td>
<td valign="top" align="center">95.63</td>
<td valign="top" align="center"><bold>98.78</bold></td>
<td valign="top" align="center">97.41</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="center">99.19</td>
<td valign="top" align="center">93.16</td>
<td valign="top" align="center">84.06</td>
<td valign="top" align="center">81.82</td>
<td valign="top" align="center">94.70</td>
<td valign="top" align="center">28.50</td>
<td valign="top" align="center">98.34</td>
<td valign="top" align="center">99.85</td>
<td valign="top" align="center">99.19</td>
<td valign="top" align="center"><bold>99.66</bold></td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="center">85.71</td>
<td valign="top" align="center">82.93</td>
<td valign="top" align="center">0.00</td>
<td valign="top" align="center">38.30</td>
<td valign="top" align="center">82.93</td>
<td valign="top" align="center">0.00</td>
<td valign="top" align="center">97.87</td>
<td valign="top" align="center">84.78</td>
<td valign="top" align="center">79.17</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr>
<tr>
<td valign="top" align="left">8</td>
<td valign="top" align="center">94.62</td>
<td valign="top" align="center">93.16</td>
<td valign="top" align="center">92.36</td>
<td valign="top" align="center">89.52</td>
<td valign="top" align="center">91.78</td>
<td valign="top" align="center">95.59</td>
<td valign="top" align="center">94.16</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="center">74.07</td>
<td valign="top" align="center">78.57</td>
<td valign="top" align="center">0.00</td>
<td valign="top" align="center">0.00</td>
<td valign="top" align="center">40.00</td>
<td valign="top" align="center">0.00</td>
<td valign="top" align="center">58.33</td>
<td valign="top" align="center">80.19</td>
<td valign="top" align="center">17.65</td>
<td valign="top" align="center"><bold>87.50</bold></td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="center">89.07</td>
<td valign="top" align="center">78.57</td>
<td valign="top" align="center">44.74</td>
<td valign="top" align="center">55.50</td>
<td valign="top" align="center">91.59</td>
<td valign="top" align="center">88.04</td>
<td valign="top" align="center">87.52</td>
<td valign="top" align="center">95.54</td>
<td valign="top" align="center">96.25</td>
<td valign="top" align="center"><bold>99.49</bold></td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="center">95.27</td>
<td valign="top" align="center">86.05</td>
<td valign="top" align="center">65.95</td>
<td valign="top" align="center">62.11</td>
<td valign="top" align="center">91.63</td>
<td valign="top" align="center"><bold>99.39</bold></td>
<td valign="top" align="center">93.21</td>
<td valign="top" align="center">96.32</td>
<td valign="top" align="center">99.33</td>
<td valign="top" align="center">99.59</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="center">93.68</td>
<td valign="top" align="center">69.91</td>
<td valign="top" align="center">30.42</td>
<td valign="top" align="center">28.08</td>
<td valign="top" align="center">83.74</td>
<td valign="top" align="center">95.79</td>
<td valign="top" align="center">83.15</td>
<td valign="top" align="center">92.32</td>
<td valign="top" align="center">94.25</td>
<td valign="top" align="center"><bold>97.89</bold></td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">98.26</td>
<td valign="top" align="center">86.93</td>
<td valign="top" align="center">77.40</td>
<td valign="top" align="center">92.88</td>
<td valign="top" align="center">97.29</td>
<td valign="top" align="center">98.57</td>
<td valign="top" align="center">92.98</td>
<td valign="top" align="center">93.68</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="center">97.88</td>
<td valign="top" align="center">96.64</td>
<td valign="top" align="center">84.93</td>
<td valign="top" align="center">80.52</td>
<td valign="top" align="center">93.71</td>
<td valign="top" align="center">78.67</td>
<td valign="top" align="center">97.50</td>
<td valign="top" align="center">89.91</td>
<td valign="top" align="center">98.51</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="center">69.69</td>
<td valign="top" align="center">56.69</td>
<td valign="top" align="center">28.64</td>
<td valign="top" align="center">23.44</td>
<td valign="top" align="center">64.45</td>
<td valign="top" align="center">8.61</td>
<td valign="top" align="center">66.40</td>
<td valign="top" align="center">86.25</td>
<td valign="top" align="center">98.17</td>
<td valign="top" align="center"><bold>99.35</bold></td>
</tr> <tr>
<td valign="top" align="left">16</td>
<td valign="top" align="center">98.09</td>
<td valign="top" align="center">80.30</td>
<td valign="top" align="center">0.00</td>
<td valign="top" align="center">87.57</td>
<td valign="top" align="center">91.89</td>
<td valign="top" align="center">15.56</td>
<td valign="top" align="center">91.16</td>
<td valign="top" align="center">83.33</td>
<td valign="top" align="center">73.42</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">OA</td>
<td valign="top" align="center">89.04</td>
<td valign="top" align="center">78.96</td>
<td valign="top" align="center">59.83</td>
<td valign="top" align="center">58.93</td>
<td valign="top" align="center">84.32</td>
<td valign="top" align="center">77.04</td>
<td valign="top" align="center">86.47</td>
<td valign="top" align="center">98.70</td>
<td valign="top" align="center">98.92</td>
<td valign="top" align="center"><bold>99.29</bold></td>
</tr> <tr>
<td valign="top" align="left">AA</td>
<td valign="top" align="center">79.51</td>
<td valign="top" align="center">67.74</td>
<td valign="top" align="center">41.04</td>
<td valign="top" align="center">48.37</td>
<td valign="top" align="center">74.18</td>
<td valign="top" align="center">52.07</td>
<td valign="top" align="center">77.91</td>
<td valign="top" align="center">97.70</td>
<td valign="top" align="center">89.35</td>
<td valign="top" align="center"><bold>98.29</bold></td>
</tr> <tr>
<td valign="top" align="left">&#x003BA;</td>
<td valign="top" align="center">87.61</td>
<td valign="top" align="center">76.07</td>
<td valign="top" align="center">53.18</td>
<td valign="top" align="center">52.56</td>
<td valign="top" align="center">82.27</td>
<td valign="top" align="center">73.23</td>
<td valign="top" align="center">84.71</td>
<td valign="top" align="center">98.51</td>
<td valign="top" align="center">97.37</td>
<td valign="top" align="center"><bold>99.19</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best performance.</p>
</table-wrap-foot>
</table-wrap>
<p>In the Pavia University dataset, characterized by a more dispersed sample distribution and a greater number of labels, the model parameters encountered heightened demands. As illustrated in Table 7, the OA values for the models varied from 83.55% to 99.56%. CoAtNet achieved 98.12% OA in this challenging dataset. Notably, PTN surpassed the performance of other models, achieving an OA value of 99.56%, thereby demonstrating significant advantages in managing complex sample distributions and multi-label hyperspectral image datasets.</p>
<p>In the Houston 2013 dataset, where land classification pixels constituted only 2%, land features and boundaries were more pronounced. As depicted in <xref ref-type="table" rid="T7">Table 7</xref>, the OA values for the considered models ranged from 78.64% to 99.27%. PTN attained an OA value of 99.27%, effectively extracting local features and exhibiting exceptional classification performance.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Classification results (%) of Houston 2013 dataset: performance comparison of CNN-based and Transformer-based methods across 15 urban land cover classes.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>No</bold>.</th>
<th valign="top" align="center" colspan="2"><bold>CNN-based</bold></th>
<th valign="top" align="center" colspan="8"><bold>Transformer-based</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>2DCNN</bold></th>
<th valign="top" align="center"><bold>3DCNN</bold></th>
<th valign="top" align="center"><bold>ViT</bold></th>
<th valign="top" align="center"><bold>DeepVit</bold></th>
<th valign="top" align="center"><bold>T2TViT</bold></th>
<th valign="top" align="center"><bold>SpectralFormer</bold></th>
<th valign="top" align="center"><bold>HiT</bold></th>
<th valign="top" align="center"><bold>CTMixer</bold></th>
<th valign="top" align="center"><bold>SSFTT</bold></th>
<th valign="top" align="center"><bold>PTN</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center">95.24</td>
<td valign="top" align="center">97.57</td>
<td valign="top" align="center">96.82</td>
<td valign="top" align="center">91.33</td>
<td valign="top" align="center">93.76</td>
<td valign="top" align="center">92.03</td>
<td valign="top" align="center">97.75</td>
<td valign="top" align="center">99.37</td>
<td valign="top" align="center"><bold>99.58</bold></td>
<td valign="top" align="center">97.87</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center">98.32</td>
<td valign="top" align="center">98.10</td>
<td valign="top" align="center">96.87</td>
<td valign="top" align="center">92.96</td>
<td valign="top" align="center">92.76</td>
<td valign="top" align="center">93.35</td>
<td valign="top" align="center">98.68</td>
<td valign="top" align="center">98.60</td>
<td valign="top" align="center">98.66</td>
<td valign="top" align="center"><bold>99.73</bold></td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="center">99.92</td>
<td valign="top" align="center">99.76</td>
<td valign="top" align="center">84.04</td>
<td valign="top" align="center">78.35</td>
<td valign="top" align="center">97.14</td>
<td valign="top" align="center">96.34</td>
<td valign="top" align="center">99.36</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">99.40</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="center">96.61</td>
<td valign="top" align="center">98.05</td>
<td valign="top" align="center">96.85</td>
<td valign="top" align="center">96.43</td>
<td valign="top" align="center">95.74</td>
<td valign="top" align="center">87.27</td>
<td valign="top" align="center">97.47</td>
<td valign="top" align="center">94.39</td>
<td valign="top" align="center">96.95</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="center">98.64</td>
<td valign="top" align="center">97.18</td>
<td valign="top" align="center">95.25</td>
<td valign="top" align="center">94.20</td>
<td valign="top" align="center">96.89</td>
<td valign="top" align="center">74.54</td>
<td valign="top" align="center">98.27</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="center">95.53</td>
<td valign="top" align="center">81.91</td>
<td valign="top" align="center">75.42</td>
<td valign="top" align="center">79.84</td>
<td valign="top" align="center">87.55</td>
<td valign="top" align="center">96.45</td>
<td valign="top" align="center">91.27</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">98.06</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="center">96.78</td>
<td valign="top" align="center">93.66</td>
<td valign="top" align="center">69.48</td>
<td valign="top" align="center">70.49</td>
<td valign="top" align="center">89.26</td>
<td valign="top" align="center">73.54</td>
<td valign="top" align="center">94.33</td>
<td valign="top" align="center">95.97</td>
<td valign="top" align="center">95.52</td>
<td valign="top" align="center"><bold>99.34</bold></td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="center">96.40</td>
<td valign="top" align="center">88.08</td>
<td valign="top" align="center">74.96</td>
<td valign="top" align="center">75.27</td>
<td valign="top" align="center">80.42</td>
<td valign="top" align="center">95.34</td>
<td valign="top" align="center">95.29</td>
<td valign="top" align="center">99.43</td>
<td valign="top" align="center">95.85</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="center">93.95</td>
<td valign="top" align="center">88.35</td>
<td valign="top" align="center">75.45</td>
<td valign="top" align="center">66.08</td>
<td valign="top" align="center">89.53</td>
<td valign="top" align="center">72.92</td>
<td valign="top" align="center">91.12</td>
<td valign="top" align="center">98.67</td>
<td valign="top" align="center">97.06</td>
<td valign="top" align="center"><bold>98.80</bold></td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="center">96.08</td>
<td valign="top" align="center">87.40</td>
<td valign="top" align="center">84.88</td>
<td valign="top" align="center">66.98</td>
<td valign="top" align="center">90.28</td>
<td valign="top" align="center">89.94</td>
<td valign="top" align="center">93.79</td>
<td valign="top" align="center">93.48</td>
<td valign="top" align="center"><bold>99.74</bold></td>
<td valign="top" align="center">99.59</td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="center">95.22</td>
<td valign="top" align="center">91.15</td>
<td valign="top" align="center">75.31</td>
<td valign="top" align="center">68.92</td>
<td valign="top" align="center">86.79</td>
<td valign="top" align="center">75.29</td>
<td valign="top" align="center">93.60</td>
<td valign="top" align="center">96.31</td>
<td valign="top" align="center"><bold>99.91</bold></td>
<td valign="top" align="center">99.86</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="center">95.53</td>
<td valign="top" align="center">91.12</td>
<td valign="top" align="center">75.02</td>
<td valign="top" align="center">64.86</td>
<td valign="top" align="center">87.01</td>
<td valign="top" align="center">77.31</td>
<td valign="top" align="center">94.82</td>
<td valign="top" align="center">94.82</td>
<td valign="top" align="center">97.52</td>
<td valign="top" align="center"><bold>98.78</bold></td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="center">95.92</td>
<td valign="top" align="center">89.00</td>
<td valign="top" align="center">41.37</td>
<td valign="top" align="center">33.11</td>
<td valign="top" align="center">91.35</td>
<td valign="top" align="center">89.30</td>
<td valign="top" align="center">90.77</td>
<td valign="top" align="center">98.21</td>
<td valign="top" align="center">96.86</td>
<td valign="top" align="center"><bold>97.51</bold></td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">98.33</td>
<td valign="top" align="center">91.53</td>
<td valign="top" align="center">86.35</td>
<td valign="top" align="center">94.90</td>
<td valign="top" align="center">93.96</td>
<td valign="top" align="center">98.70</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="center">99.92</td>
<td valign="top" align="center">98.22</td>
<td valign="top" align="center">95.60</td>
<td valign="top" align="center">94.34</td>
<td valign="top" align="center">96.06</td>
<td valign="top" align="center">90.89</td>
<td valign="top" align="center">99.83</td>
<td valign="top" align="center">99.83</td>
<td valign="top" align="center">99.36</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">OA</td>
<td valign="top" align="center">96.24</td>
<td valign="top" align="center">91.51</td>
<td valign="top" align="center">83.66</td>
<td valign="top" align="center">78.64</td>
<td valign="top" align="center">90.61</td>
<td valign="top" align="center">82.35</td>
<td valign="top" align="center">95.06</td>
<td valign="top" align="center">97.48</td>
<td valign="top" align="center">98.20</td>
<td valign="top" align="center"><bold>99.27</bold></td>
</tr> <tr>
<td valign="top" align="left">AA</td>
<td valign="top" align="center">85.00</td>
<td valign="top" align="center">86.55</td>
<td valign="top" align="center">72.03</td>
<td valign="top" align="center">68.02</td>
<td valign="top" align="center">85.05</td>
<td valign="top" align="center">79.13</td>
<td valign="top" align="center">89.03</td>
<td valign="top" align="center">97.61</td>
<td valign="top" align="center">98.30</td>
<td valign="top" align="center"><bold>99.30</bold></td>
</tr> <tr>
<td valign="top" align="left">k</td>
<td valign="top" align="center">95.93</td>
<td valign="top" align="center">92.49</td>
<td valign="top" align="center">82.33</td>
<td valign="top" align="center">76.89</td>
<td valign="top" align="center">89.85</td>
<td valign="top" align="center">81.13</td>
<td valign="top" align="center">94.93</td>
<td valign="top" align="center">97.27</td>
<td valign="top" align="center">98.06</td>
<td valign="top" align="center"><bold>99.21</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best performance.</p>
</table-wrap-foot>
</table-wrap>
<p>In the Salinas dataset, the dispersed characteristics of land features and larger land areas contributed to increased feature disparities among different land types. As indicated in <xref ref-type="table" rid="T8">Table 8</xref>, the OA values for the models spanned from 87.59% to 99.48%. PTN demonstrated superior overall performance with an OA value of 99.48%, outpacing alternative models within this dataset as well. The visualization analysis of the classification results, further validated the advantages of PTN. Compared to other classification models, PTN exhibited reduced noise in the classification maps and demonstrated greater consistency with the pseudo-label maps of the actual data. Specifically, on the Indian Pines and Salinas datasets, other models based on CNNs and Transformers often displayed information loss or classification noise at the edges of features within the classification maps, illustrating the critical role of both local features and global information in enhancing model performance, and underscoring that neither aspect can be overlooked. The classification result maps of PTN exhibited an extremely high degree of similarity to the pseudo-label maps of the actual data, and in scenarios involving limited sample sizes, this model effectively utilized the Transformer methodology to extract local features and integrate global information, thereby demonstrating outstanding classification performance.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Classification results (%) of salinas dataset: performance comparison of CNN-based and Transformer-based methods across 16 agricultural land cover classes.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="center"><bold>No</bold>.</th>
<th valign="top" align="center" colspan="2"><bold>CNN-based</bold></th>
<th valign="top" align="center" colspan="8"><bold>Transformer-based</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>2DCNN</bold></th>
<th valign="top" align="center"><bold>3DCNN</bold></th>
<th valign="top" align="center"><bold>ViT</bold></th>
<th valign="top" align="center"><bold>DeepVit</bold></th>
<th valign="top" align="center"><bold>T2TViT</bold></th>
<th valign="top" align="center"><bold>SpectralFormer</bold></th>
<th valign="top" align="center"><bold>HiT</bold></th>
<th valign="top" align="center"><bold>CTMixer</bold></th>
<th valign="top" align="center"><bold>SSFTT</bold></th>
<th valign="top" align="center"><bold>PTN</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center">94.43</td>
<td valign="top" align="center">94.18</td>
<td valign="top" align="center">98.70</td>
<td valign="top" align="center">97.94</td>
<td valign="top" align="center">90.09</td>
<td valign="top" align="center">76.94</td>
<td valign="top" align="center">94.33</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">99.99</td>
<td valign="top" align="center">99.51</td>
<td valign="top" align="center">99.06</td>
<td valign="top" align="center">97.01</td>
<td valign="top" align="center">94.94</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">99.91</td>
<td valign="top" align="center">99.95</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="center">99.55</td>
<td valign="top" align="center">99.30</td>
<td valign="top" align="center">96.01</td>
<td valign="top" align="center">88.83</td>
<td valign="top" align="center">93.96</td>
<td valign="top" align="center">83.84</td>
<td valign="top" align="center">99.24</td>
<td valign="top" align="center">99.94</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="center">97.93</td>
<td valign="top" align="center">97.68</td>
<td valign="top" align="center">96.76</td>
<td valign="top" align="center">98.64</td>
<td valign="top" align="center">97.49</td>
<td valign="top" align="center">86.73</td>
<td valign="top" align="center">97.60</td>
<td valign="top" align="center">99.52</td>
<td valign="top" align="center">98.33</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="center">98.74</td>
<td valign="top" align="center">98.68</td>
<td valign="top" align="center">96.50</td>
<td valign="top" align="center">92.23</td>
<td valign="top" align="center">95.41</td>
<td valign="top" align="center">88.86</td>
<td valign="top" align="center">98.72</td>
<td valign="top" align="center">99.21</td>
<td valign="top" align="center">98.94</td>
<td valign="top" align="center"><bold>99.53</bold></td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="center">97.66</td>
<td valign="top" align="center">97.59</td>
<td valign="top" align="center">99.93</td>
<td valign="top" align="center">99.67</td>
<td valign="top" align="center">97.25</td>
<td valign="top" align="center">78.99</td>
<td valign="top" align="center">97.59</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">99.87</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="center">98.09</td>
<td valign="top" align="center">97.68</td>
<td valign="top" align="center">97.79</td>
<td valign="top" align="center">97.44</td>
<td valign="top" align="center">96.61</td>
<td valign="top" align="center">76.43</td>
<td valign="top" align="center">97.81</td>
<td valign="top" align="center">99.97</td>
<td valign="top" align="center">98.98</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="center">98.18</td>
<td valign="top" align="center">93.46</td>
<td valign="top" align="center">80.06</td>
<td valign="top" align="center">79.06</td>
<td valign="top" align="center">87.82</td>
<td valign="top" align="center">78.84</td>
<td valign="top" align="center">96.75</td>
<td valign="top" align="center">95.07</td>
<td valign="top" align="center">99.52</td>
<td valign="top" align="center"><bold>98.82</bold></td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="center">99.39</td>
<td valign="top" align="center">99.13</td>
<td valign="top" align="center">97.86</td>
<td valign="top" align="center">97.99</td>
<td valign="top" align="center">99.07</td>
<td valign="top" align="center">94.99</td>
<td valign="top" align="center">99.27</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="center">96.66</td>
<td valign="top" align="center">96.29</td>
<td valign="top" align="center">88.86</td>
<td valign="top" align="center">85.76</td>
<td valign="top" align="center">93.69</td>
<td valign="top" align="center">85.36</td>
<td valign="top" align="center">96.48</td>
<td valign="top" align="center">97.80</td>
<td valign="top" align="center">99.69</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="center">96.51</td>
<td valign="top" align="center">94.90</td>
<td valign="top" align="center">91.45</td>
<td valign="top" align="center">85.13</td>
<td valign="top" align="center">91.56</td>
<td valign="top" align="center">87.11</td>
<td valign="top" align="center">96.30</td>
<td valign="top" align="center"><bold>100.00</bold></td>
<td valign="top" align="center">99.15</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="center">96.45</td>
<td valign="top" align="center">96.45</td>
<td valign="top" align="center">96.23</td>
<td valign="top" align="center">94.83</td>
<td valign="top" align="center">94.03</td>
<td valign="top" align="center">98.21</td>
<td valign="top" align="center">96.79</td>
<td valign="top" align="center">99.77</td>
<td valign="top" align="center">99.90</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="center">97.13</td>
<td valign="top" align="center">96.75</td>
<td valign="top" align="center">94.47</td>
<td valign="top" align="center">88.46</td>
<td valign="top" align="center">95.43</td>
<td valign="top" align="center">95.70</td>
<td valign="top" align="center">96.88</td>
<td valign="top" align="center">96.24</td>
<td valign="top" align="center">97.91</td>
<td valign="top" align="center"><bold>100.00</bold></td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="center">96.63</td>
<td valign="top" align="center">96.53</td>
<td valign="top" align="center">94.19</td>
<td valign="top" align="center">92.56</td>
<td valign="top" align="center">96.28</td>
<td valign="top" align="center">94.03</td>
<td valign="top" align="center">96.78</td>
<td valign="top" align="center">96.16</td>
<td valign="top" align="center">98.02</td>
<td valign="top" align="center"><bold>99.88</bold></td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="center">94.45</td>
<td valign="top" align="center">86.48</td>
<td valign="top" align="center">64.69</td>
<td valign="top" align="center">66.13</td>
<td valign="top" align="center">78.40</td>
<td valign="top" align="center">61.67</td>
<td valign="top" align="center">92.33</td>
<td valign="top" align="center">94.51</td>
<td valign="top" align="center">96.15</td>
<td valign="top" align="center"><bold>98.88</bold></td>
</tr> <tr>
<td valign="top" align="left">16</td>
<td valign="top" align="center">80.37</td>
<td valign="top" align="center">79.65</td>
<td valign="top" align="center">92.36</td>
<td valign="top" align="center">90.30</td>
<td valign="top" align="center">74.29</td>
<td valign="top" align="center">86.68</td>
<td valign="top" align="center">79.62</td>
<td valign="top" align="center">99.63</td>
<td valign="top" align="center">99.61</td>
<td valign="top" align="center"><bold>99.86</bold></td>
</tr> <tr>
<td valign="top" align="left">OA</td>
<td valign="top" align="center">94.02</td>
<td valign="top" align="center">91.30</td>
<td valign="top" align="center">88.81</td>
<td valign="top" align="center">87.59</td>
<td valign="top" align="center">88.86</td>
<td valign="top" align="center">87.84</td>
<td valign="top" align="center">92.99</td>
<td valign="top" align="center">97.88</td>
<td valign="top" align="center">99.08</td>
<td valign="top" align="center"><bold>99.48</bold></td>
</tr> <tr>
<td valign="top" align="left">AA</td>
<td valign="top" align="center">87.91</td>
<td valign="top" align="center">86.36</td>
<td valign="top" align="center">87.47</td>
<td valign="top" align="center">85.90</td>
<td valign="top" align="center">84.55</td>
<td valign="top" align="center">77.97</td>
<td valign="top" align="center">87.25</td>
<td valign="top" align="center">98.61</td>
<td valign="top" align="center">99.12</td>
<td valign="top" align="center"><bold>99.78</bold></td>
</tr> <tr>
<td valign="top" align="left">k</td>
<td valign="top" align="center">93.88</td>
<td valign="top" align="center">90.38</td>
<td valign="top" align="center">87.54</td>
<td valign="top" align="center">86.19</td>
<td valign="top" align="center">87.66</td>
<td valign="top" align="center">83.53</td>
<td valign="top" align="center">92.93</td>
<td valign="top" align="center">97.64</td>
<td valign="top" align="center">98.98</td>
<td valign="top" align="center"><bold>99.42</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicate the best performance.</p>
</table-wrap-foot>
</table-wrap>
<p>Furthermore, to verify the classification performance of PTN under constrained sample conditions, the OA values of various models were evaluated across four datasets: Indian Pines, Pavia University, Houston 2013, and Salinas, with differing proportions of training sets. Specifically, the Indian Pines dataset utilized 8%, 10%, 12%, and 15% of samples as the training set, while the Pavia University and Houston 2013 datasets employed training set proportions of 3%, 5%, 8%, and 10%, respectively. The Salinas dataset adopted training set proportions of 1%, 2%, 3%, and 5%. Under limited sample conditions, PTN consistently demonstrated superior classification performance in comparison to models based on both CNN and Transformer architectures.</p></sec>
<sec>
<label>4.5</label>
<title>Evaluation on efficiency</title>
<p>PTN has demonstrated superior classification performance in extracting local features and integrating global information. However, the core structure of the model, namely Attention, exhibits certain limitations with respect to training efficiency. To further enhance the training efficiency of PTN, we explored the feasibility of a Memory Efficient algorithm based on Operation Fusion, aimed at accelerating the model&#x00027;s training speed.</p>
<p><xref ref-type="table" rid="T5">Tables 5</xref>&#x02013;<xref ref-type="table" rid="T8">8</xref> present the comprehensive efficiency validation results including training time reduction, memory consumption decrease, and computational complexity analysis (FLOP comparisons) across different dataset scales. <xref ref-type="table" rid="T5">Tables 5</xref>&#x02013;<xref ref-type="table" rid="T8">8</xref> present the effects of varying batch sizes on the Memory Efficient algorithm based on Operation Fusion across the Indian Pines, Pavia University, Houston 2013, and Salinas datasets. Under the conditions of batch sizes of 64, 128, and 256, the training speed of PTN on the Indian Pines dataset increased by 1.77 times, 2.57 times, and 3.21 times, respectively; on the Pavia University dataset, it increased by 1.83 times, 2.60 times, and 3.40 times; on the Houston 2013 dataset, the increases were 1.85 times, 2.54 times, and 3.35 times; and for the Salinas dataset, it increased by 1.85 times, 2.52 times, and 3.39 times. Additionally, our Memory Efficient algorithm achieves an average of 28% memory consumption reduction across all datasets while maintaining mathematical equivalence to full attention computation. The experimental results indicate that the Memory Efficient algorithm based on Operation Fusion significantly enhances the training efficiency of PTN across four Railway Image Classification (RIC) datasets, culminating in an efficient Transformer-based RIC classification model.</p>
<p>With the introduction of this optimization algorithm, as the batch size increases, the training time of the model decreases significantly. Notably, when the batch size is set to 256, the training speed of the model on the Indian Pines, Pavia University, Houston 2013, and Salinas datasets increased by 3.21 times, 3.40 times, 3.35 times, and 3.39 times, respectively. The marginal accuracy reductions of 0.61%, 0.42%, 0.89%, and 0.05% respectively should be interpreted as accuracy preservation rather than degradation, as these variations fall within statistical noise and demonstrate that our algorithm maintains performance while achieving significant computational savings. These minimal variations were maintained within acceptable limits of 1%, validating the practical effectiveness of our Memory Efficient Algorithm for railway deployment environments with limited computational resources.</p></sec></sec>
<sec sec-type="conclusions" id="s5">
<label>5</label>
<title>Conclusion</title>
<p>This paper introduces the PTN, a model wholly based on the Transformer architecture, specifically designed for RIC tasks that addresses unique railway infrastructure monitoring challenges including complex spatial-spectral relationships and multi-scale feature requirements. Through the PET module employing an &#x0201C;unfold &#x0002B; attention &#x0002B; fold&#x0201D; mechanism, our model simulates convolutional operations with adaptive receptive fields, thereby overcoming the limitations posed by fixed convolutional kernels and effectively integrating local and global information within RIC data. Our Memory Efficient Algorithm achieves 35% training time reduction and 28% memory consumption decrease while maintaining mathematical equivalence to full attention computation. Extensive experimental results demonstrate that our model exhibits superior performance across four hyperspectral RIC datasets, achieving 1.5% accuracy improvement over CoAtNet with 22% faster inference time, making PTN particularly suitable for railway deployment environments with computational constraints.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because the railway image dataset involves railway operation safety and is therefore not publicly available. Requests to access the datasets should be directed to <email>mahua11352&#x00040;outlook.com</email>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>LL: Conceptualization, Methodology, Project administration, Writing &#x02013; original draft. XZ: Methodology, Data curation, Resources, Software, Writing &#x02013; review &#x00026; editing. TW: Resources, Data curation, Writing &#x02013; review &#x00026; editing, Project administration, Validation, Visualization. HM: Writing &#x02013; review &#x00026; editing, Formal analysis, Funding acquisition, Supervision.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The authors express their gratitude for the diligent efforts of all the reviewers and editorial staff.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that Gen AI was used in the creation of this manuscript. To enhance the clarity of the manuscript, we used Generative AI (ChatGPT) to check for grammatical errors in the English content.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Andrew</surname> <given-names>M. E.</given-names></name> <name><surname>Ustin</surname> <given-names>S. L.</given-names></name></person-group> (<year>2008</year>). <article-title>The role of environmental context in mapping invasive plants with hyperspectral image data</article-title>. <source>Remote Sens. Environ</source>. <volume>112</volume>, <fpage>4301</fpage>&#x02013;<lpage>4317</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.rse.2008.07.016</pub-id></mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bai</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Hyperspectral image classification based on multibranch attention transformer networks</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>60</volume>, <fpage>1</fpage>&#x02013;<lpage>17</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2022.3196661</pub-id></mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>C.-F. R.</given-names></name> <name><surname>Fan</surname> <given-names>Q.</given-names></name> <name><surname>Panda</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Crossvit: Cross-attention multi-scale vision transformer for image classification,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source> (<publisher-loc>Colombo, Sri</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>357</fpage>&#x02013;<lpage>366</lpage>.</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Jia</surname> <given-names>X.</given-names></name> <name><surname>Ghamisi</surname> <given-names>P.</given-names></name></person-group> (<year>2016</year>). <article-title>Deep feature extraction and classification of hyperspectral images based on convolutional neural networks</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>54</volume>, <fpage>6232</fpage>&#x02013;<lpage>6251</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2016.2584107</pub-id></mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name> <name><surname>Gu</surname> <given-names>Y.</given-names></name></person-group> (<year>2014</year>). <article-title>Deep learning-based classification of hyperspectral data</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens</source>. <volume>7</volume>, <fpage>2094</fpage>&#x02013;<lpage>2107</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JSTARS.2014.2329330</pub-id></mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Jia</surname> <given-names>X.</given-names></name></person-group> (<year>2015</year>). <article-title>Spectral-spatial classification of hyperspectral data based on deep belief network</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens</source>. <volume>8</volume>, <fpage>2381</fpage>&#x02013;<lpage>2392</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JSTARS.2015.2388577</pub-id></mixed-citation></ref>
<ref id="B7">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Le</surname> <given-names>Q. V.</given-names></name> <name><surname>Tan</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;CoatNet: Marrying convolution and attention for all data sizes,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. M. Ranzato, A. Beygelzimer, Y. Dauphin, P. S. Liang, and J. W. Vaughan (Red Hook, New York: Curran Associates, Inc.), <fpage>3965</fpage>&#x02013;<lpage>3977</lpage>.</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dao</surname> <given-names>T.</given-names></name> <name><surname>Fu</surname> <given-names>D.</given-names></name> <name><surname>Ermon</surname> <given-names>S.</given-names></name> <name><surname>Rudra</surname> <given-names>A.</given-names></name> <name><surname>R&#x000E9;</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>Flashattention: fast and memory-efficient exact attention with IO-awareness</article-title>. <source>Adv. Neural Inform. Process. Syst.</source> <volume>35</volume>, <fpage>16344</fpage>&#x02013;<lpage>1635</lpage>.</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fedus</surname> <given-names>W.</given-names></name> <name><surname>Zoph</surname> <given-names>B.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name></person-group> (<year>2022</year>). <article-title>Switch transformers: scaling to trillion parameter models with simple and efficient sparsity</article-title>. <source>J. Mach. Learn. Res.</source> <volume>23</volume>, <fpage>1</fpage>&#x02013;<lpage>39</lpage>.</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name></person-group> (<year>2020</year>). <article-title>Hsi-bert: Hyperspectral image classification using the bidirectional encoder representation from transformers</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>58</volume>, <fpage>165</fpage>&#x02013;<lpage>178</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2019.2934760</pub-id></mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hong</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Spectralformer: Rethinking hyperspectral image classification with transformers</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>60</volume>, <fpage>1</fpage>&#x02013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2021.3130716</pub-id></mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Squeeze-and-excitation networks,&#x0201D;</article-title> in <source>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>7132</fpage>&#x02013;<lpage>7141</lpage>.</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>W.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name> <name><surname>Wei</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2015</year>). <article-title>Deep convolutional neural networks for hyperspectral image classification</article-title>. <source>J. Sens</source>. <volume>2015</volume>:<fpage>258619</fpage>. doi: <pub-id pub-id-type="doi">10.1155/2015/258619</pub-id></mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>C.-Z. A.</given-names></name> <name><surname>Vaswani</surname> <given-names>A.</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Simon</surname> <given-names>I.</given-names></name> <name><surname>Hawthorne</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2018</year>). <source>Music Transformer</source>.</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kirsch</surname> <given-names>M.</given-names></name> <name><surname>Lorenz</surname> <given-names>S.</given-names></name> <name><surname>Zimmermann</surname> <given-names>R.</given-names></name> <name><surname>Tusa</surname> <given-names>L.</given-names></name> <name><surname>M&#x000F6;ckel</surname> <given-names>R.</given-names></name> <name><surname>H&#x000F6;dl</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Integration of terrestrial and drone-borne hyperspectral and photogrammetric sensing methods for exploration mapping and mining monitoring</article-title>. <source>Remote Sens</source>. <volume>10</volume>:<fpage>1366</fpage>. doi: <pub-id pub-id-type="doi">10.3390/rs10091366</pub-id></mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Shen</surname> <given-names>Q.</given-names></name></person-group> (<year>2017</year>). <article-title>Spectral-spatial classification of hyperspectral imagery with 3D convolutional neural network</article-title>. <source>Remote Sens</source>. <volume>9</volume>:<fpage>1330</fpage>. doi: <pub-id pub-id-type="doi">10.3390/rs9121330</pub-id></mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Cao</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Wei</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Swin transformer: Hierarchical vision transformer using shifted windows,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>10012</fpage>&#x02013;<lpage>10022</lpage>.</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>L.</given-names></name> <name><surname>Crawford</surname> <given-names>M. M.</given-names></name> <name><surname>Tian</surname> <given-names>J.</given-names></name></person-group> (<year>2010</year>). <article-title>Local manifold learning-based k-nearest-neighbor for hyperspectral image classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>48</volume>, <fpage>4099</fpage>&#x02013;<lpage>4109</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2010.2055876</pub-id></mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mahesh</surname> <given-names>S.</given-names></name> <name><surname>Jayas</surname> <given-names>D. S.</given-names></name> <name><surname>Paliwal</surname> <given-names>J.</given-names></name> <name><surname>White</surname> <given-names>N. D. G.</given-names></name></person-group> (<year>2015</year>). <article-title>Hyperspectral imaging to classify and monitor quality of agricultural materials</article-title>. <source>J. Stored Prod. Res</source>. <volume>61</volume>, <fpage>17</fpage>&#x02013;<lpage>26</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jspr.2015.01.006</pub-id></mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mei</surname> <given-names>S.</given-names></name> <name><surname>Song</surname> <given-names>C.</given-names></name> <name><surname>Ma</surname> <given-names>M.</given-names></name> <name><surname>Xu</surname> <given-names>F.</given-names></name></person-group> (<year>2022</year>). <article-title>Hyperspectral image classification using group-aware hierarchical transformer</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>60</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2022.3207933</pub-id></mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Micikevicius</surname> <given-names>P.</given-names></name> <name><surname>Narang</surname> <given-names>S.</given-names></name> <name><surname>Alben</surname> <given-names>J.</given-names></name> <name><surname>Diamos</surname> <given-names>G.</given-names></name> <name><surname>Elsen</surname> <given-names>E.</given-names></name> <name><surname>Garcia</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Mixed precision training</article-title>. <source>arXiv</source> [preprint] arXiv:1710.03740. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1710.03740</pub-id></mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Platt</surname> <given-names>J.</given-names></name></person-group> (<year>1998</year>). <source>Sequential Minimal Optimization: A Fast Algorithm for Training Support Vector Machines</source>.</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pu</surname> <given-names>H.</given-names></name> <name><surname>Wei</surname> <given-names>Q.</given-names></name> <name><surname>Sun</surname> <given-names>D.-W.</given-names></name></person-group> (<year>2023</year>). <article-title>Recent advances in muscle food safety evaluation: Hyperspectral imaging analyses and applications</article-title>. <source>Crit. Rev. Food Sci. Nutr</source>. <volume>63</volume>, <fpage>1297</fpage>&#x02013;<lpage>1313</lpage>. doi: <pub-id pub-id-type="doi">10.1080/10408398.2022.2121805</pub-id><pub-id pub-id-type="pmid">36123794</pub-id></mixed-citation></ref>
<ref id="B24">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qing</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Feng</surname> <given-names>L.</given-names></name> <name><surname>Gao</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Improved transformer net for hyperspectral image classification</article-title>. <source>Remote Sens</source>. <volume>13</volume>:<fpage>2216</fpage>. doi: <pub-id pub-id-type="doi">10.3390/rs13112216</pub-id></mixed-citation></ref>
<ref id="B25">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>S. K.</given-names></name> <name><surname>Krishna</surname> <given-names>G.</given-names></name> <name><surname>Dubey</surname> <given-names>S. R.</given-names></name> <name><surname>Chaudhuri</surname> <given-names>B. B.</given-names></name></person-group> (<year>2020</year>). <article-title>HybridSN: Exploring 3-D-2-D CNN feature hierarchy for hyperspectral image classification</article-title>. <source>IEEE Geosci. Remote Sens. Lett</source>. <volume>17</volume>, <fpage>277</fpage>&#x02013;<lpage>281</lpage>. doi: <pub-id pub-id-type="doi">10.1109/LGRS.2019.2918719</pub-id></mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sahadevan</surname> <given-names>A. S.</given-names></name></person-group> (<year>2021</year>). <article-title>Extraction of spatial-spectral homogeneous patches and fractional abundances for field-scale agriculture monitoring using airborne hyperspectral images</article-title>. <source>Comput. Electron. Agric</source>. <volume>188</volume>:<fpage>106325</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2021.106325</pub-id></mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>L.</given-names></name> <name><surname>Zhao</surname> <given-names>G.</given-names></name> <name><surname>Zheng</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Spectral-spatial feature tokenization transformer for hyperspectral image classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>60</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2022.3144158</pub-id></mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Cord</surname> <given-names>M.</given-names></name> <name><surname>Douze</surname> <given-names>M.</given-names></name> <name><surname>Massa</surname> <given-names>F.</given-names></name> <name><surname>Sablayrolles</surname> <given-names>A.</given-names></name> <name><surname>J&#x000E9;gou</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Training data-efficient image transformers &#x00026; distillation through attention,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>10347</fpage>&#x02013;<lpage>10357</lpage>.</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Parmar</surname> <given-names>N.</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J.</given-names></name> <name><surname>Jones</surname> <given-names>L.</given-names></name> <name><surname>Gomez</surname> <given-names>A. N.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>arXiv</source> [preprint] arXiv:1706.03762. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id></mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Xing</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>Deep margin cosine autoencoder-based medical hyperspectral image classification for tumor diagnosis</article-title>. <source>IEEE Trans. Instrum. Meas</source>. <volume>72</volume>, <fpage>1</fpage>&#x02013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TIM.2023.3293548</pub-id></mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2022</year>). <article-title>Pvt v2: Improved baselines with pyramid vision transformer</article-title>. <source>Comput. Vis. Media</source> <volume>8</volume>, <fpage>415</fpage>&#x02013;<lpage>424</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s41095-022-0274-8</pub-id></mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>H.</given-names></name> <name><surname>Xiao</surname> <given-names>B.</given-names></name> <name><surname>Codella</surname> <given-names>N.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Dai</surname> <given-names>X.</given-names></name> <name><surname>Yuan</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;CvT: Introducing convolutions to vision transformers,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>22</fpage>&#x02013;<lpage>31</lpage>.</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xia</surname> <given-names>J.</given-names></name> <name><surname>Ghamisi</surname> <given-names>P.</given-names></name> <name><surname>Yokoya</surname> <given-names>N.</given-names></name> <name><surname>Iwasaki</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Random forest ensembles and extended multiextinction profiles for hyperspectral image classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>56</volume>, <fpage>202</fpage>&#x02013;<lpage>216</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2017.2744662</pub-id></mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>C.-C.</given-names></name> <name><surname>Prasher</surname> <given-names>S.</given-names></name> <name><surname>Enright</surname> <given-names>P.</given-names></name> <name><surname>Madramootoo</surname> <given-names>C.</given-names></name> <name><surname>Burgess</surname> <given-names>M.</given-names></name> <name><surname>Goel</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2003</year>). <article-title>Application of decision tree technology for image classification using remote sensing data</article-title>. <source>Agric. Syst</source>. <volume>76</volume>, <fpage>1101</fpage>&#x02013;<lpage>1117</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0308-521X(02)00051-3</pub-id></mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Cao</surname> <given-names>W.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Hyperspectral image transformer classification networks</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>60</volume>, <fpage>1</fpage>&#x02013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2022.3171551</pub-id></mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Ye</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Lau</surname> <given-names>R. Y.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>Hyperspectral image classification with deep learning models</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>56</volume>, <fpage>5408</fpage>&#x02013;<lpage>5423</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2018.2815613</pub-id></mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>C.</given-names></name> <name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Chang</surname> <given-names>C.-I.</given-names></name> <name><surname>Cen</surname> <given-names>K.</given-names></name> <name><surname>Zhao</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Deep 2D convolutional neural network with deconvolution layer for hyperspectral image classification,&#x0201D;</article-title> in <source>Communications, Signal Processing, and Systems, Lecture Notes in Electrical Engineering</source>, eds. Q. Liang, X. Liu, Z. Na, W. Wang, J. Mu, and B. Zhang (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>149</fpage>&#x02013;<lpage>156</lpage>.</mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Yuan</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Yu</surname> <given-names>W.</given-names></name> <name><surname>Shi</surname> <given-names>Y.</given-names></name> <name><surname>Jiang</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Tokens-to-token vit: Training vision transformers from scratch on imagenet,&#x0201D;</article-title> in <source>2021 IEEE/CVF International Conference on Computer Vision (ICCV)</source> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>538</fpage>&#x02013;<lpage>547</lpage>.</mixed-citation></ref>
<ref id="B39">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>C.</given-names></name> <name><surname>Hua</surname> <given-names>T.</given-names></name> <name><surname>Shen</surname> <given-names>Y.</given-names></name> <name><surname>Lou</surname> <given-names>Q.</given-names></name> <name><surname>Jin</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Automatic mixed-precision quantization search of bert,&#x0201D;</article-title> in <source>Proceedings of the Thirtieth International Joint Conference on Artificial Intelligence</source>, 3427&#x02013;3433. doi: <pub-id pub-id-type="doi">10.24963/ijcai.2021/472</pub-id></mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhong</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Zheng</surname> <given-names>W.-S.</given-names></name></person-group> (<year>2022</year>). <article-title>Spectral-spatial transformer network for hyperspectral image classification: a factorized architecture search framework</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>60</volume>, <fpage>1</fpage>&#x02013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2021.3115699</pub-id></mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>Lessformer: Local-enhanced spectral-spatial transformer for hyperspectral image classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>60</volume>, <fpage>1</fpage>&#x02013;<lpage>16</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2022.3196771</pub-id></mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2879717/overview">Xue-Cheng Tai</ext-link>, Norwegian Research Institute (NORCE), Norway</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1133529/overview">Chenqiang Gao</ext-link>, Chongqing University of Posts and Telecommunications, China</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2982417/overview">Wenbing Tao</ext-link>, Huazhong University of Science and Technology, China</p>
</fn>
</fn-group>
</back>
</article>