<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Sci.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Computer Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Sci.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2624-9898</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcomp.2025.1613648</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Improving remote sensing scene classification with data augmentation techniques to mitigate class imbalance</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Wang</surname> <given-names>Ping</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Zhao</surname> <given-names>Xin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/3029855"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Yuanhui</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhan</surname> <given-names>Lili</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Qingdao Huanghai University</institution>, <city>Qingdao</city>, <country country="cn">China</country></aff>
<aff id="aff2"><label>2</label><institution>College of Geodesy and Geomatics, Shandong University of Science and Technology</institution>, <city>Qingdao</city>, <country country="cn">China</country></aff>
<author-notes>
<corresp id="c001"><label>&#x0002A;</label>Correspondence: Lili Zhan, <email xlink:href="mailto:skd992016@sdust.edu.cn">skd992016@sdust.edu.cn</email></corresp>
<fn fn-type="equal" id="fn001"><label>&#x02020;</label><p>These authors have contributed equally to this work and share first authorship</p></fn></author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-10-08">
<day>08</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>7</volume>
<elocation-id>1613648</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Wang, Zhao, Chen and Zhan.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, Zhao, Chen and Zhan</copyright-holder>
<license>
<ali:license_ref start_date="2025-10-08">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>High-resolution remote sensing imagery is a powerful tool that provides massive information about ground objects. However, conventional methods often fail to achieve satisfactory results for complex urban scene classification. This is attributed to the fact that conventional methods are unable to meet the requirements of high-accuracy remote sensing image scene classification (RSSC) and are hindered by challenges such as limited labeled samples and class imbalance, which may lead to classification bias in classifiers. On the contrary, deep learning-based RSSC represents an important approach for understanding semantic information. This paper explores the feasibility of mitigating classification bias by reducing the imbalance ratio (IR) of the dataset. First, a class-imbalanced dataset was constructed using very high-resolution (VHR) images, labeled into nine land use/land cover (LULC) categories. Second, comprehensive data augmentation techniques (mirroring, rotation, cropping, Hue, Saturation, Value (HSV) perturbation, and gamma transformation) were applied, successfully reducing the dataset&#x00027;s IR from 9.38 to 1.28. Subsequently, four architectures, MobileNet-v2, ResNet101, ResNeXt101_32 &#x000D7; 32d, and Transformer, were trained and evaluated on both class-balanced and class-imbalanced datasets. The results indicate that the classification bias caused by class imbalance was alleviated, significantly improving the classifier&#x00027;s performance. Specifically for the most severely underrepresented category (intersections), precision and recall improvements reached up to 128% and 102%, respectively, narrowing the gap with other categories and reducing classification bias. Furthermore, the average Kappa and overall accuracy (OA) increased by 11.84% and 12.97%, respectively, with reduced standard deviations in recall and precision, demonstrating enhanced model stability.</p></abstract>
<kwd-group>
<kwd>remote sensing image scene classification</kwd>
<kwd>class-imbalance</kwd>
<kwd>deep learning</kwd>
<kwd>fine-tune</kwd>
<kwd>data augment</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This article has been supported by the funds of the National Natural Science Foundation of China (41971339) and the State Key Laboratory of Synthetical Automation for Process Industries (Northeastern University) [SAPI-2024-KFKT-08].</funding-statement>
</funding-group>
<counts>
<fig-count count="10"/>
<table-count count="8"/>
<equation-count count="6"/>
<ref-count count="41"/>
<page-count count="14"/>
<word-count count="7300"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computer Vision</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>High spatial resolution remote sensing imagery encompasses multiple low-level features (e.g., spectral characteristics, patterns, shadows, and textures) and explicit high-level semantic features. High-resolution remote sensing scene classification (RSSC) is essential for environmental understanding and has gained significant attention in remote sensing image interpretation (<xref ref-type="bibr" rid="B12">Gu et al., 2019</xref>; <xref ref-type="bibr" rid="B3">Cheng et al., 2017</xref>, <xref ref-type="bibr" rid="B5">2020</xref>). Scene classification involves categorizing images into predefined semantic classes based on scene-specific information, representing high-level abstractions of scene content. However, traditional classification methods are limited by their reliance on low-level feature analysis, which restricts their capacity to extract high-level semantic information, thereby often failing to meet the accuracy demands of RSSC. Currently, the proliferation of deep learning has spurred numerous methodologies for remote sensing scene image classification, which can be broadly categorized into three types: autoencoder-based, convolutional neural network (CNN)-based, and generative adversarial network (GAN)-based approaches (<xref ref-type="bibr" rid="B3">Cheng et al., 2017</xref>; <xref ref-type="bibr" rid="B27">Ma et al., 2019</xref>; <xref ref-type="bibr" rid="B38">Yu et al., 2020</xref>; <xref ref-type="bibr" rid="B4">Cheng et al., 2022</xref>). For instance, <xref ref-type="bibr" rid="B8">Du et al. (2017</xref>) proposed stacked convolutional denoising auto-encoders, which demonstrated superior classification performance compared to state-of-the-art unsupervised networks. <xref ref-type="bibr" rid="B33">Tang et al. (2021</xref>) proposed the attention consistent network (ACNet) based on Siamese networks and validated the method using three remote sensing scene datasets, demonstrating that the proposed method achieves good performance. <xref ref-type="bibr" rid="B4">Cheng et al. (2022</xref>) proposed perturbation-seeking generative adversarial networks (PSGANs) to improve the stability of sample generation. Recent advances have also incorporated attention mechanisms and multiscale feature fusion, with notable contributions including co-enhanced global-part integration approaches (<xref ref-type="bibr" rid="B40">Zhao et al., 2024a</xref>), gradient-guided multiscale focal attention networks (<xref ref-type="bibr" rid="B41">Zhao et al., 2024b</xref>), and multibranch fusion-based feature enhancement methods (<xref ref-type="bibr" rid="B25">Liu et al., 2019</xref>), which have demonstrated remarkable performance improvements.</p>
<p>Despite these extensive advances in RSSC, the challenges posed by limited labeled samples and class imbalance remain insufficiently addressed. Deep learning, as a data-driven paradigm, relies heavily on data availability and quality, computational resources, architectural innovations, and optimization techniques. Among these factors, dataset quality is a critical determinant of model performance, with class balance being a key quality indicator. When training datasets are imbalanced, classifiers tend to favor majority classes and often fail to correctly identify minority classes (<xref ref-type="bibr" rid="B2">Buda et al., 2018</xref>; <xref ref-type="bibr" rid="B18">Johnson and Khoshgoftaar, 2019</xref>; <xref ref-type="bibr" rid="B21">Leevy et al., 2018</xref>; <xref ref-type="bibr" rid="B26">Luque et al., 2019</xref>; <xref ref-type="bibr" rid="B36">Yessou et al., 2020</xref>; <xref ref-type="bibr" rid="B34">Thabtah et al., 2020</xref>). In extreme cases, particularly with noise-robust models, minority classes may even be completely ignored, leading to severe classification bias.</p>
<p>Approaches to addressing class imbalance in machine learning are typically divided into two categories: data-level methods and algorithm-level methods. Data-level methods aim to modify the distribution of training samples to improve the effectiveness of standard algorithms (<xref ref-type="bibr" rid="B11">Fern&#x000E1;ndez et al., 2018</xref>). For instance, <xref ref-type="bibr" rid="B24">Liu and Tsoumakas (2020</xref>) used random under-sampling to rebalance class distribution. Among over-sampling techniques, the synthetic minority oversampling technique (SMOTE) has been recognized as one of the most influential data-level strategies (<xref ref-type="bibr" rid="B7">Douzas et al., 2018</xref>). SMOTE is fundamentally a clustering-based method designed to oversample one-dimensional vector data&#x02014;such as spectral vectors in multispectral or hyperspectral images (<xref ref-type="bibr" rid="B10">Feng et al., 2019</xref>)&#x02014;to alleviate overfitting. However, its effectiveness is limited when applied to two-dimensional or three-dimensional image data due to its tendency to introduce noise. In contrast, algorithm-level methods maintain the original data distribution and instead focus on adjusting training strategies or inference mechanisms. For example, <xref ref-type="bibr" rid="B1">Bria et al. (2020</xref>) proposed a two-stage deep learning framework to address severe class imbalance in small lesion detection. <xref ref-type="bibr" rid="B29">Ren et al. (2020</xref>) introduced an improved DeepLab V3&#x0002B; for remote sensing image segmentation, incorporating a loss function tailored to sample distribution. In addition, few-shot learning approaches offer alternative perspectives on handling limited minority class samples. Recent studies by <xref ref-type="bibr" rid="B6">Deng et al. (2024</xref>) on masked second-order pooling for few-shot RSSC demonstrates the potential of meta-learning approaches. Fine-tuning strategies have also been shown to enhance performance under imbalanced conditions, with <xref ref-type="bibr" rid="B13">Guan et al. (2020</xref>) developing a random fine-tuning meta metric learning (RF-MML) model for aerial image classification, which demonstrated effectiveness in handling imbalanced data. Despite these advancements, the effectiveness of class imbalance mitigation strategies, particularly data augmentation approaches, in deep learning-based RSSC tasks remains systematically underexplored.</p>
<p>While high-resolution remote sensing images contain rich semantic information, their scene classification faces significant challenges from limited labeled samples and class imbalance. Although some studies have investigated class imbalance in remote sensing contexts (<xref ref-type="bibr" rid="B13">Guan et al., 2020</xref>), comprehensive research in this area remains limited. This study aims to address these gaps through a systematic investigation of class-imbalanced RSSC. The specific objectives of this study are:</p>
<p>&#x02022; Evaluating the sensitivity of overall accuracy metrics in RSSC to class imbalance</p>
<p>&#x02022; Investigating the integration of data augmentation methods with datasets and examining the feasibility of alleviating classification bias through data augmentation methods</p>
<p>&#x02022; Analyzing the sensitivity of individual categories in RSSC to class imbalance</p>
<p>&#x02022; Assessing the sensitivity of RSSC to different classification algorithms</p>
<p><xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the experimental workflow. First, a class-imbalanced high-resolution remote sensing scene image dataset is constructed. Second, comprehensive data augmentation methods are applied according to dataset characteristics, reducing the dataset&#x00027;s imbalance ratio (IR) by approximately one-seventh. Subsequently, the classifiers are trained and fine-tuned on both augmented and original datasets. Finally, model performance on overall and individual categories is evaluated, with visualization of results from selected study areas.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Workflow of this study.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0001.tif">
<alt-text content-type="machine-generated">Flowchart depicting a machine learning workflow. A scene image feeds into an original dataset, IR equals 9.38, and is augmented 10.68 times creating an augmented dataset, IR equals 1.28. Both datasets input into pre-trained models: ResNet101, MobileNet V2, ResNeXt32*32d, and Transformers, which are fine-tuned for 200 epochs, resulting in trained models. These models are used to predict, leading to evaluation and visualization outputs.</alt-text>
</graphic>
</fig>
<p>The main contributions of this paper include investigating the feasibility of alleviating the class-imbalance problems using data augmentation methods in RSSC and advancing research in this field. In addition, a class-imbalanced RSSC benchmark dataset is constructed, and deep learning methods with different parameters and structures are evaluated.</p>
<p>As for the organization of this article, Section 2 introduces the methodology and experimental design in detail. Sections 3 and 4 present and discuss the experimental results comprehensively. Finally, Section 5 summarizes the conclusions and outlines future research directions.</p></sec>
<sec sec-type="materials|methods" id="s2">
<label>2</label>
<title>Materials and methods</title>
<sec>
<label>2.1</label>
<title>Study region</title>
<p>Shinan district is located in the southern area of Qingdao, Shandong province, characterized by a compact north&#x02013;south extent of 4.5 km and an elongated east-west span of 12.7 km, with a total area of 30.01 km<sup>2</sup> and approximately 15 km of coastline. The study region encompasses Shinan district and its surrounding areas (including marine regions), ranging from 120.2793229&#x000B0;E to 120.4292515&#x000B0;E and from 36.0376721&#x000B0;N to 36.0973352&#x000B0;N (<xref ref-type="fig" rid="F2">Figure 2</xref>). Based on the data from the National Geomatics Center of China, the 30-m LULC of Shinan District and its surrounding areas is presented as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Combining <xref ref-type="fig" rid="F2">Figures 2</xref>, <xref ref-type="fig" rid="F3">3</xref>, the main LULC of the study region includes water bodies (not including the ocean), artificial surfaces, shrubland, cultivated land, grassland, and bare land. The artificial surface category exhibits complex sub-classifications, encompassing buildings of different heights, roads, and open-air venues. This diversity makes the region particularly suitable for RSSC research. Furthermore, as the study region represents a typical urban built-up area, accurate scene classification is essential for understanding urban spatial distribution patterns through semantic analysis of scene images.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>VHR and the location of the region of interest.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0002.tif">
<alt-text content-type="machine-generated">Map displaying a region with marked cities including ChengYang, LiCang, ShiBei, and HuangDao, alongside a zoomed-in satellite view of Shinan. Coordinates and a scale in miles are provided, with a compass rose indicating cardinal directions.</alt-text>
</graphic>
</fig>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Thirty-m LULC map of Shinan District and its surrounding areas (including the marine region).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0003.tif">
<alt-text content-type="machine-generated">Map of land use categories with color coding: dark red for artificial surfaces, beige for bare land, gray for cultivated land, light green for grassland, green for shrubland, and blue for water bodies.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<label>2.2</label>
<title>Dataset</title>
<p>The very high-resolution (VHR) image used in this study was downloaded from data processed by LocaSpace Viewer (<ext-link ext-link-type="uri" xlink:href="http://www.tuxingis.com/locaspace.html">http://www.tuxingis.com/locaspace.html</ext-link>), with a size of 56,064<sup>&#x0002A;</sup>27,648 pixels and a spatial resolution of 0.2985 m. According to information obtained from the <ext-link ext-link-type="uri" xlink:href="https://resources.maxar.com">https://resources.maxar.com</ext-link>, the original source consists of WorldView-3 data (<xref ref-type="table" rid="T1">Table 1</xref>), acquired on 16 March 2019, with a maximum ground sample distance of 0.36 m, a sun elevation of 49.9&#x000B0;, an image off-nadir angle of 23.3&#x000B0;, and a maximum target azimuth of 5.1&#x000B0;. The full archive of this data is available through ESA (<ext-link ext-link-type="uri" xlink:href="https://earth.esa.int/eogateway/&#x0007E;catalog/worldview-3-full-archive-and-tasking">https://earth.esa.int/eogateway/&#x0007E;catalog/worldview-3-full-archive-and-tasking</ext-link>).</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Introduction to the original data source.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Item</bold></th>
<th valign="top" align="center"><bold>Info</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Time</td>
<td valign="top" align="center">16 March 2019</td>
</tr> <tr>
<td valign="top" align="left">Max GSD</td>
<td valign="top" align="center">0.36 m</td>
</tr> <tr>
<td valign="top" align="left">Sun elevation</td>
<td valign="top" align="center">49.9&#x000B0;</td>
</tr> <tr>
<td valign="top" align="left">Max target azimuth</td>
<td valign="top" align="center">5.1&#x000B0;</td>
</tr> <tr>
<td valign="top" align="left">Image of nadir</td>
<td valign="top" align="center">23.3&#x000B0;</td>
</tr></tbody>
</table>
</table-wrap>
<p>The VHR image was segmented into 256<sup>&#x0002A;</sup>256 pixel scene images, with each scene image covering a ground area of 76.416 &#x000D7; 76.416 m<sup>2</sup>. A total of 23,652 (108 &#x000D7; 219) scene images were generated. Based on visual interpretation and the spatial characteristics of the study area (<xref ref-type="fig" rid="F2">Figures 2</xref>, <xref ref-type="fig" rid="F3">3</xref>), the images were divided into nine predefined semantic classes: chaparral, dense buildings, high-rise sparse buildings, low-rise sparse buildings, intersections, open-air venues, roads, water (including the ocean), and water&#x02013;land junctions. Subsequently, 8% of the scene images were selected and manually labeled as training and validation samples to initialize the dataset (<xref ref-type="fig" rid="F4">Figure 4</xref>).</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Examples of each class are shown, where C1&#x02013;C9 represent chaparral, dense buildings, high-rise sparse buildings, low-rise sparse buildings, intersections, open-air venues, roads, water, and coastline, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0004.tif">
<alt-text content-type="machine-generated">Grid of satellite images labeled C1 to C9. C1 shows a dense forest. C2 depicts residential buildings and roads. C3 displays tall apartment buildings. C4 and C7 show urban street intersections. C5 includes densely packed apartment buildings. C6 captures a circular park area. C8 appears mostly dark, possibly a night shot or dark terrain. C9 shows a rocky coastline with waves.</alt-text>
</graphic>
</fig>
<p>As detailed in <xref ref-type="table" rid="T2">Table 2</xref>, the number of samples in the C4 (intersections) category is the smallest (68 samples), while the number of samples in the C8 (water) category is the largest (638 samples). The remaining categories contain samples ranging from approximately 100&#x02013;250 samples each.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Sample number of each class in the original dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>No</bold>.</th>
<th valign="top" align="center"><bold>Classes</bold></th>
<th valign="top" align="center" colspan="2"><bold>Number</bold></th>
</tr>
<tr>
<th/>
<th/>
<th valign="top" align="center"><bold>Original</bold></th>
<th valign="top" align="center"><bold>Augmented</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Class 1</td>
<td valign="top" align="center">Chaparral</td>
<td valign="top" align="center">256 (13.8%)</td>
<td valign="top" align="center">2,256 (11.4%)</td>
</tr> <tr>
<td valign="top" align="left">Class 2</td>
<td valign="top" align="center">Dense buildings</td>
<td valign="top" align="center">148 (8%)</td>
<td valign="top" align="center">2,148 (10.8%)</td>
</tr> <tr>
<td valign="top" align="left">Class 3</td>
<td valign="top" align="center">High-rise sparse buildings</td>
<td valign="top" align="center">94 (5.1%)</td>
<td valign="top" align="center">2,094 (10.5%)</td>
</tr> <tr>
<td valign="top" align="left">Class 4</td>
<td valign="top" align="center">Intersections</td>
<td valign="top" align="center">68 (3.7%)</td>
<td valign="top" align="center">2,068 (10.4%)</td>
</tr> <tr>
<td valign="top" align="left">Class 5</td>
<td valign="top" align="center">Low-rise sparse buildings</td>
<td valign="top" align="center">206 (11.1%)</td>
<td valign="top" align="center">2,206 (11.1%)</td>
</tr> <tr>
<td valign="top" align="left">Class 6</td>
<td valign="top" align="center">Open-air venues</td>
<td valign="top" align="center">122 (6.6%)</td>
<td valign="top" align="center">2,122 (10.7%)</td>
</tr> <tr>
<td valign="top" align="left">Class 7</td>
<td valign="top" align="center">Roads</td>
<td valign="top" align="center">212 (11.4%)</td>
<td valign="top" align="center">2,212 (11.1%)</td>
</tr> <tr>
<td valign="top" align="left">Class 8</td>
<td valign="top" align="center">Water</td>
<td valign="top" align="center">638 (34.3%)</td>
<td valign="top" align="center">2,638 (13.3%)</td>
</tr> <tr>
<td valign="top" align="left">Class 9</td>
<td valign="top" align="center">Water&#x02013;land junction</td>
<td valign="top" align="center">114 (6.1%)</td>
<td valign="top" align="center">2,114 (10.6%)</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<label>2.3</label>
<title>Data augmentation</title>
<p>To address class imbalance issues, data augmentation techniques were implemented (<xref ref-type="table" rid="T3">Table 3</xref>). The objective was to generate 2,000 augmented samples per category. The quotient derived from dividing the target number by the sample count of a given category determined the base multiple for sample expansion, while samples corresponding to the remainder received an incremented multiple. This process facilitated the calculation of each sample&#x00027;s expansion factor. For instance, the water category, with 638 samples, yielded a quotient of 3 and a remainder of 86 when 2,000 was divided by 638. Consequently, the base multiple for the water category was set to 3, with the 86 randomly selected samples assigned a multiple of 4.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Data augmentation method.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>Formula</bold></th>
<th valign="top" align="center"><bold>Probability</bold></th>
<th valign="top" align="center"><bold>Scope</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Mirroring</td>
<td valign="top" align="center">\</td>
<td valign="top" align="center">0.5</td>
<td valign="top" align="center">\</td>
</tr> <tr>
<td valign="top" align="left">Cropping</td>
<td valign="top" align="center">\</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.8</td>
</tr> <tr>
<td valign="top" align="left">Rotation</td>
<td valign="top" align="center"><inline-formula><mml:math id="M1"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>u</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>v</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>=<inline-formula><mml:math id="M2"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mtd><mml:mtd><mml:mo>-</mml:mo><mml:mo class="qopname">sin</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>&#x003B8;</mml:mi></mml:mtd><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>x</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>y</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">1</td>
<td valign="top" align="center"><inline-formula><mml:math id="M3"><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mn>30</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>&#x02264;</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mn>30</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>C</mml:mi><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mn>3</mml:mn></mml:mstyle></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mn>180</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>&#x02264;</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mn>180</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>s</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></inline-formula></td>
</tr> <tr>
<td valign="top" align="left">HSV perturbation</td>
<td valign="top" align="center"><italic>Max, Min</italic> &#x0003D; max(<italic>R, G, B</italic>), min(<italic>R, G, B</italic>) <italic>H</italic>=<inline-formula><mml:math id="M4"><mml:mi>H</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msup><mml:mn>0</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x000A0;</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mn>60</mml:mn></mml:mrow><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>&#x000A0;</mml:mo><mml:mo>*</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>G</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>R</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>+</mml:mo><mml:msup><mml:mn>60</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mi>B</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mn>60</mml:mn></mml:mrow><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>&#x000A0;</mml:mo><mml:mo>*</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>B</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>+</mml:mo><mml:msup><mml:mn>180</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mi>R</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mn>60</mml:mn></mml:mrow><mml:mo>&#x02022;</mml:mo></mml:msup><mml:mo>&#x000A0;</mml:mo><mml:mo>*</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>R</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>B</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>+</mml:mo><mml:msup><mml:mn>300</mml:mn><mml:mo>&#x02218;</mml:mo></mml:msup><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mi>G</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="top" align="left"><italic>S</italic> = <italic>Max</italic>&#x02212;<italic>Min</italic> </td>
</tr>
<tr>
<td valign="top" align="left"><italic>V</italic> = <italic>Max</italic> </td>
</tr>
<tr>
<td valign="top" align="left"><italic>C</italic> = <italic>S</italic> </td>
</tr>
<tr>
<td valign="top" align="left"><italic>H</italic>&#x02032;=<inline-formula><mml:math id="M5"><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02022;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mn>60</mml:mn></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="top" align="left"><italic>X</italic> = <italic>C</italic><sup>&#x0002A;</sup>(1&#x02212;|<italic>H</italic>&#x02032; <italic>mod</italic> 2 &#x02212; 1|)</td>
</tr>
<tr>
<td valign="top" align="left">(<italic>R, G, B</italic>) =<inline-formula><mml:math id="M6"><mml:mo>&#x000A0;</mml:mo><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mrow><mml:mi>V</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>C</mml:mi></mml:mrow><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>*</mml:mo></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>u</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>&#x0003C;</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>&#x0003C;</mml:mo><mml:mn>2</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>X</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>&#x0003C;</mml:mo><mml:mn>3</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mn>3</mml:mn><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>&#x0003C;</mml:mo><mml:mn>4</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mn>4</mml:mn><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>&#x0003C;</mml:mo><mml:mn>5</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo stretchy='false'>(</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>X</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mn>5</mml:mn><mml:mo>&#x02264;</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo>&#x0003C;</mml:mo><mml:mn>6</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">&#x02212;10&#x000B0; &#x02264; &#x025B3;<italic>H</italic> &#x02264; 10&#x000B0; &#x02013;0.1<italic>S</italic> &#x02264; &#x025B3;<italic>S</italic> &#x02264; 0.1<italic>S</italic> &#x02212;0.1<italic>V</italic> &#x02264; &#x025B3;<italic>V</italic> &#x02264; 0.1<italic>V</italic> </td>
</tr> <tr>
<td valign="top" align="left">Gamma transformation</td>
<td valign="top" align="center"><italic>s</italic> &#x0003D; <italic>cr</italic><sup>&#x003B3;</sup></td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.5 &#x02264; &#x003B3; &#x02264; 2</td>
</tr></tbody>
</table>
</table-wrap>
<p>Subsequently, augmentation methods were applied to each sample based on its calculated multiple (<xref ref-type="fig" rid="F5">Figure 5</xref>), encompassing mirroring, rotation, cropping, hue, saturation, and value (HSV) perturbation, and gamma transformation (<xref ref-type="bibr" rid="B31">Shi et al., 2020</xref>; <xref ref-type="bibr" rid="B22">Lewis et al., 1989</xref>). Augmentation parameters were determined through empirical validation on subsets of each class to ensure semantic consistency and optimal performance. Specifically, the mirroring probability was set to 0.5, while all other techniques were applied with a probability of 1.0. For rotation operations, class-specific parameter ranges were implemented: C3 (high-rise sparse buildings) was limited to &#x000B1;30&#x000B0; to preserve architectural orientation and geometric relationships, while other classes utilized the full &#x000B1;180&#x000B0; range. During rotation, additional cropping was used to eliminate edge artifacts. Given the slight deviation of the central axis from the vertical in bird&#x00027;s-eye view imagery, the perturbation ranges for HSV components were adjusted according to class characteristics: C8 (water) received reduced perturbation ranges (&#x000B1;5&#x000B0;, &#x000B1;5%, &#x000B1;5%) to maintain its distinctive spectral signature, while other classes used standard ranges of &#x000B1;10&#x000B0;, &#x000B1;10%, and &#x000B1;10% for H, S, and V components, respectively. For structured categories such as C4 (intersections) and C7 (roads), stricter cropping boundaries were enforced to prevent semantic category shifts during spatial augmentation. The &#x003B3; value for the gamma transformation was set in the range 0.5&#x02013;2. Due to the size alterations induced by augmentation operations, a resizing step was incorporated to ensure compatibility with the classifier&#x00027;s input requirements. Finally, the augmented dataset was constructed by merging the augmented samples of each category with the original dataset (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Workflow of data augmentation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0005.tif">
<alt-text content-type="machine-generated">Flowchart of image processing steps starting with mirroring and cropping. A decision node checks for Class C3, leading to rotation options of plus thirty degrees or plus or minus one hundred eighty degrees. It continues with HSV perturbation, converting between RGB and HSV, applying adjustments, gamma transformation, and finally resizing.</alt-text>
</graphic>
</fig>
<p><xref ref-type="fig" rid="F6">Figure 6</xref> presents examples of augmented samples, where (a) represents the original image, while (b), (c), and (d) show augmented samples exhibiting richer color, contrast, and spatial features. The augmentation results appear realistic and consistent with satellite imagery characteristics.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Examples of augmentation effects. <bold>(a)</bold> represents the original image, while <bold>(b&#x02013;d)</bold> show augmented samples.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0006.tif">
<alt-text content-type="machine-generated">Four aerial images labeled (a), (b), (c), and (d) show a cityscape with buildings and streets. Each image appears progressively sharper, displaying a clearer view of the same urban area, highlighting structures and urban details more distinctly.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<label>2.4</label>
<title>Imbalance ratio</title>
<p>The imbalance ratio (IR) is defined as the ratio of the number of samples in the majority class to that in the minority class, as expressed in <xref ref-type="disp-formula" rid="EQ1">Equation 1</xref>.</p>
<disp-formula id="EQ1"><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>I</mml:mi><mml:mi>m</mml:mi><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>R</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>j</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(1)</label></disp-formula>
<p>where <bold>C</bold><sub><bold>majority</bold></sub> is the number of samples in the majority class and <bold>C</bold><sub><bold>minority</bold></sub> is the number of samples in the minority class.</p>
<p>The IR of the original dataset was 9.38 (628/68), indicating significant class imbalance. This imbalance issue was substantially alleviated in the augmented dataset, achieving an IR of 1.28 (2,638/2,068) (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
</sec>
<sec>
<label>2.5</label>
<title>Methods</title>
<p>Four models were used for scene classification in this study and compared, including Mobilenet-v2, ResNet101, resnextt101_32x32d, and Transformer (<xref ref-type="table" rid="T4">Table 4</xref>). A brief introduction to each model architecture is provided below.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>File size of used models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Name</bold></th>
<th valign="top" align="center"><bold>File size (MB)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">MobileNet-v2</td>
<td valign="top" align="center">8.83</td>
</tr> <tr>
<td valign="top" align="left">ResNet101</td>
<td valign="top" align="center">162</td>
</tr> <tr>
<td valign="top" align="left">ResNeXt101_32<sup>&#x0002A;</sup>32d</td>
<td valign="top" align="center">1,740</td>
</tr> <tr>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="center">313</td>
</tr></tbody>
</table>
</table-wrap>
<sec>
<label>2.5.1</label>
<title>MobileNet</title>
<p>Fundamentally, MobileNet-v1 (<xref ref-type="bibr" rid="B14">Harjoseputro et al., 2020</xref>) was developed by replacing the standard convolutional layers in Visual Geometry Group (VGG) with the depth-wise separable convolution layers. Specifically, ordinary convolutions are decomposed into a depth-wise convolution and a point-wise convolution, enabling similar performance with reduced computational cost. In addition, Rectified Linear Unit 6 (ReLU6) replaces Rectified Linear Unit (ReLU), thereby limiting the activation function values to a specified boundary. Furthermore, the residual structures and the squeeze and excitation (SE) modules (<xref ref-type="bibr" rid="B17">Hu et al., 2018</xref>) were introduced into MobileNet-v2 (<xref ref-type="bibr" rid="B30">Sandler et al., 2018</xref>) and MobileNet-v3 (<xref ref-type="bibr" rid="B16">Howard et al., 2019</xref>), respectively.</p></sec>
<sec>
<label>2.5.2</label>
<title>ResNet</title>
<p>The development of ResNet (<xref ref-type="bibr" rid="B15">He et al., 2016</xref>) challenged the conventional belief that &#x0201C;the deeper the network, the higher the accuracy rate.&#x0201D; Experimental evidence demonstrated that network accuracy initially improves with increasing depth, but after reaching a saturation point, performance drops sharply. This phenomenon is attributed to the inability of very deep networks to perform &#x0201C;identity transformations (y = x)&#x0201D; due to overly strong non-linear transformation capabilities. To address this issue, shortcut connections were added to the ResNet block (<xref ref-type="fig" rid="F7">Figure 7a</xref>), with 1<sup>&#x0002A;</sup>1 convolutions integrated into the main branches of the down-sampling block to achieve a balance between linear and non-linear conversion.</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Structure of <bold>(a)</bold> ResNet block, <bold>(b)</bold> ResNeXt block, and <bold>(c)</bold> Inception block.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0007.tif">
<alt-text content-type="machine-generated">Diagrams of three neural network architectures: (a) ResNet Block shows three convolutional layers with a shortcut connection. (b) ResNeXt Block features multiple parallel paths, each with convolutional layers combining outputs. (c) Inception Block displays parallel convolutional operations of various sizes, merging outputs at the end.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<label>2.5.3</label>
<title>ResNeXt</title>
<p>ResNeXt (<xref ref-type="bibr" rid="B35">Xie et al., 2017</xref>) simplifies the &#x0201C;split&#x02013;transform&#x02013;merge&#x0201D; structure of the Inception block (<xref ref-type="bibr" rid="B32">Szegedy et al., 2015</xref>) (<xref ref-type="fig" rid="F7">Figure 7c</xref>) into a uniform topological structure. By combining this approach with Resnet&#x00027;s shortcut connections (<xref ref-type="bibr" rid="B15">He et al., 2016</xref>), ResNeXt introduces grouped convolution, which represents an intermediate architecture between ordinary convolution and depth-wise separable convolution, achieving improved computational efficiency. When the dimension of the input feature map is 256, the ResNeXt block uses 32 groups of independent convolutions with identical topology (<xref ref-type="fig" rid="F7">Figure 7b</xref>).</p></sec>
<sec>
<label>2.5.4</label>
<title>Transformer</title>
<p>Vision transformer (ViT-B/16) represents a paradigm shift from convolutional architectures to transformer-based approaches for image classification. Unlike CNNs that process images through hierarchical feature extraction, ViT divides input images into fixed-size patches (16 &#x000D7; 16 pixels), treats them as sequences, and applies standard transformer architectures originally designed for natural language processing. The ViT-B/16 model consists of 12 transformer layers with 12 attention heads and a hidden dimension of 768. Each image patch is linearly embedded and combined with positional encodings before being fed into the transformer encoder. This architecture enables the model to capture long-range dependencies and global context through self-attention mechanisms, which can be particularly beneficial for RSSC, where spatial relationships across the entire scene are crucial for accurate classification.</p>
</sec>
</sec>
<sec>
<label>2.6</label>
<title>Fine-tuning</title>
<p>In numerous computer vision applications, the performance of deep learning models diminishes considerably when confronted with a scarcity of labeled data. Fine-tuning pre-trained models offers a straightforward and efficient transfer learning approach, enabling generalization to novel tasks with limited training samples, accelerating model convergence, and mitigating overfitting (<xref ref-type="bibr" rid="B37">Yosinski et al., 2014</xref>; <xref ref-type="bibr" rid="B39">Zhao et al., 2025</xref>).</p>
<p>Specifically, feature extractors trained on the source domain can be effectively transferred to the target domain through fine-tuning. By utilizing an extractor pre-trained on a large dataset (the source domain), the parameters of the trained feature extractor can be repurposed to initialize the classifier for training on a new dataset (the target domain). In this study, pre-trained models were used to initialize the model parameters, and the output layer (the number of categories) was modified to align with the specific requirements of the dataset.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<label>3</label>
<title>Results</title>
<sec>
<label>3.1</label>
<title>Experiment</title>
<p>The experiments were performed on a server configured with Ubuntu 16.04, a GPU (RTX2080Ti), and the Cuda10.2 &#x0002B; Python3.7 &#x0002B; Pytorch1.7 framework, as detailed in <xref ref-type="table" rid="T5">Table 5</xref>.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Experiment environment.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Item</bold></th>
<th valign="top" align="center"><bold>Version</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Operating system</td>
<td valign="top" align="center">Ubuntu 16.04</td>
</tr> <tr>
<td valign="top" align="left">GPU</td>
<td valign="top" align="center">RTX2080Ti</td>
</tr> <tr>
<td valign="top" align="left">Framework</td>
<td valign="top" align="center">Python3.7&#x0002B; Pytorch1.7&#x0002B; Cuda10.2</td>
</tr></tbody>
</table>
</table-wrap>
<p>Prior to training, the pre-trained models from ImageNet were used to initialize the model parameters, while the number of output classes was adjusted to 9 to accommodate the fine-tuning process. The optimizer was set to Adaptive Moment Estimation (Adam). A combination of gradual warm-up and MultiStepLR strategies was used for learning rate scheduling. Specifically, during the first four epochs, the learning rate &#x003B5; (10<sup>&#x02212;6</sup>) linearly increased from this small initial value to the baseline learning rate of 0.001. This approach enabled the classifiers to effectively incorporate prior knowledge while ensuring faster and more stable training. The MultiStepLR milestones were set as 70% and 90% of maximum iterations (20 epochs). Upon reaching each milestone, the learning rate was reduced by a factor of gamma (0.1) relative to the previous value.</p>
<p>To assess the efficacy of data augmentation, experiments were performed on the original imbalanced datasets. The dataset was split into training (72%), validation (8%), and testing (20%) sets. Validation results were averaged over several experimental runs to ensure statistical robustness. Given that the sample count in the class-balanced dataset had increased by a factor of 10.688, the training epochs for the comparative experiments were extended to 200 epochs to ensure fair comparison and balance the impact of different dataset sizes. This adjustment was equivalent to training the model for 10 times longer on the original class-imbalanced dataset.</p>
</sec>
<sec>
<label>3.2</label>
<title>Evaluation metrics</title>
<p>In this study, Kappa, recall, precision, and OA were used as the evaluation metrics, with the means of recall and precision values calculated for each category. Recall represents the probability that positive samples from a specific category are correctly classified, while precision represents the probability that samples predicted as belonging to a particular category are actually correctly classified. OA represents the proportion of correctly classified samples relative to the total sample count, and the Kappa measures the agreement between predicted and true labels, accounting for chance agreement. <xref ref-type="table" rid="T6">Table 6</xref> presents a typical confusion matrix, with Kappa, precision, recall, and OA computed according to <xref ref-type="disp-formula" rid="EQ2">Equations 2</xref>&#x02013;<xref ref-type="disp-formula" rid="EQ5">5</xref>.</p>
<disp-formula id="EQ2"><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(2)</label></disp-formula>
<disp-formula id="EQ3"><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(3)</label></disp-formula>
<disp-formula id="EQ4"><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>O</mml:mi><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:mi>N</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(4)</label></disp-formula>
<disp-formula id="EQ5"><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>K</mml:mi><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>N</mml:mi><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(5)</label></disp-formula>
<p>where <bold>P</bold><sub><bold>ij</bold></sub> is the number of samples whose true label is <bold>C</bold><sub><bold>i</bold></sub> and whose predicted label is <bold>C</bold><sub><bold>j</bold></sub>, and <italic>N</italic> is the total number of samples. In addition, the overall performance of results from multiple models can be characterized by mean (<xref ref-type="disp-formula" rid="EQ6">Equation 6</xref>).</p>
<disp-formula id="EQ6"><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(6)</label></disp-formula>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Typical confusion matrix.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th/>
<th/>
<th valign="top" align="center" colspan="5"><bold>Predicted label</bold></th>
</tr>
<tr>
<th/>
<th/>
<th valign="top" align="center"><bold>C</bold><sub>1</sub></th>
<th valign="top" align="center"><bold>C</bold><sub>2</sub></th>
<th valign="top" align="center"><bold>&#x02026;</bold></th>
<th valign="top" align="center"><bold>C</bold><sub>n</sub></th>
<th valign="top" align="center"><italic><bold>P</bold><sub>&#x0002B;<italic>j</italic></sub></italic></th>
</tr> 
</thead>
<tbody>
<tr>
<td valign="top" align="left">True label</td>
<td valign="top" align="center">C<sub>1</sub></td>
<td valign="top" align="center"><italic>P<sub>11</sub></italic></td>
<td valign="top" align="center"><italic>P<sub>21</sub></italic></td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center"><italic>P<sub>&#x0002B;1</sub></italic></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">C<sub>2</sub></td>
<td valign="top" align="center"><italic>P<sub>12</sub></italic></td>
<td valign="top" align="center"><italic>P<sub>22</sub></italic></td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center"><italic>P<sub>&#x0002B;2</sub></italic></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td/>
</tr>
 <tr>
<td/>
<td valign="top" align="center">C<sub>n</sub></td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center"><italic>P<sub>&#x0002B;<italic>n</italic></sub></italic></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center"><italic>P<sub><italic>i</italic>&#x0002B;</sub></italic></td>
<td valign="top" align="center"><italic>P<sub>1&#x0002B;</sub></italic></td>
<td valign="top" align="center"><italic>P<sub>2&#x0002B;</sub></italic></td>
<td valign="top" align="center">&#x02026;</td>
<td valign="top" align="center"><italic>P<sub><italic>n</italic>&#x0002B;</sub></italic></td>
<td valign="top" align="center"><italic>N</italic></td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<label>3.3</label>
<title>Experiment results and analysis</title>
<sec>
<label>3.3.1</label>
<title>Quantitative analysis</title>
<p>In this study, four models trained on class-imbalanced and class-balanced datasets were compared, i.e., mobilenet-v2, ResNet101, ResNeXt101_32<sup>&#x0002A;</sup>32d, and Transformer. <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T8">8</xref> present the experimental results on the VHR images. The following conclusions can be drawn:</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Experiment indicators of models on a class-balanced dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model category</bold></th>
<th valign="top" align="center" colspan="2"><bold>MobileNetv2</bold></th>
<th valign="top" align="center" colspan="2"><bold>ResNet101</bold></th>
<th valign="top" align="center" colspan="2"><bold>ResNeXt101_32</bold><sup><bold>&#x0002A;</bold></sup><bold>32d</bold></th>
<th valign="top" align="center" colspan="2"><bold>Transformer</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">C1</td>
<td valign="top" align="center">0.963</td>
<td valign="top" align="center">0.899</td>
<td valign="top" align="center">0.976</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.966</td>
<td valign="top" align="center">0.922</td>
<td valign="top" align="center">0.968</td>
<td valign="top" align="center">0.948</td>
</tr> <tr>
<td valign="top" align="left">C2</td>
<td valign="top" align="center">0.922</td>
<td valign="top" align="center">0.886</td>
<td valign="top" align="center">0.943</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.922</td>
<td valign="top" align="center">0.916</td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center">0.951</td>
</tr> <tr>
<td valign="top" align="left">C3</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center">0.99</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.942</td>
<td valign="top" align="center">0.99</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.963</td>
</tr> <tr>
<td valign="top" align="left">C4</td>
<td valign="top" align="center">0.907</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.948</td>
<td valign="top" align="center">0.927</td>
<td valign="top" align="center">0.968</td>
<td valign="top" align="center">0.938</td>
<td valign="top" align="center">0.955</td>
</tr> <tr>
<td valign="top" align="left">C5</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.883</td>
<td valign="top" align="center">0.926</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.922</td>
<td valign="top" align="center">0.913</td>
<td valign="top" align="center">0.944</td>
<td valign="top" align="center">0.895</td>
</tr> <tr>
<td valign="top" align="left">C6</td>
<td valign="top" align="center">0.886</td>
<td valign="top" align="center">0.959</td>
<td valign="top" align="center">0.917</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.914</td>
<td valign="top" align="center">0.973</td>
<td valign="top" align="center">0.917</td>
<td valign="top" align="center">0.943</td>
</tr> <tr>
<td valign="top" align="left">C7</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.839</td>
<td valign="top" align="center">0.932</td>
<td valign="top" align="center">0.904</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.873</td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center">0.941</td>
</tr> <tr>
<td valign="top" align="left">C8</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.995</td>
<td valign="top" align="center">0.995</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">C9</td>
<td valign="top" align="center">0.9</td>
<td valign="top" align="center">0.912</td>
<td valign="top" align="center">0.933</td>
<td valign="top" align="center">0.918</td>
<td valign="top" align="center">0.927</td>
<td valign="top" align="center">0.917</td>
<td valign="top" align="center">0.918</td>
<td valign="top" align="center">0.933</td>
</tr> <tr>
<td valign="top" align="left">Average</td>
<td valign="top" align="center">0.921</td>
<td valign="top" align="center">0.921</td>
<td valign="top" align="center">0.946</td>
<td valign="top" align="center">0.946</td>
<td valign="top" align="center">0.941</td>
<td valign="top" align="center">0.941</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.948</td>
</tr> <tr>
<td valign="top" align="left">Kappa</td>
<td valign="top" align="center" colspan="2">0.973</td>
<td valign="top" align="center" colspan="2">0.973</td>
<td valign="top" align="center" colspan="2">0.973</td>
<td valign="top" align="center" colspan="2">0.979</td>
</tr> <tr>
<td valign="top" align="left">OA</td>
<td valign="top" align="center" colspan="2">0.922</td>
<td valign="top" align="center" colspan="2">0.947</td>
<td valign="top" align="center" colspan="2">0.942</td>
<td valign="top" align="center" colspan="2">0.951</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Experiment indicators of models on the class-imbalanced dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model category</bold></th>
<th valign="top" align="center" colspan="2"><bold>MobileNetv2</bold></th>
<th valign="top" align="center" colspan="2"><bold>ResNet101</bold></th>
<th valign="top" align="center" colspan="2"><bold>ResNeXt101_32</bold><sup><bold>&#x0002A;</bold></sup><bold>32d</bold></th>
<th valign="top" align="center" colspan="2"><bold>Transformer</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">C1</td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.944</td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.944</td>
<td valign="top" align="center">0.943</td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.948</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">C2</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.944</td>
<td valign="top" align="center">0.9</td>
<td valign="top" align="center">0.947</td>
<td valign="top" align="center">0.9</td>
<td valign="top" align="center">0.857</td>
<td valign="top" align="center">0.818</td>
<td valign="top" align="center">0.948</td>
</tr> <tr>
<td valign="top" align="left">C3</td>
<td valign="top" align="center">0.846</td>
<td valign="top" align="center">0.846</td>
<td valign="top" align="center">0.923</td>
<td valign="top" align="center">0.857</td>
<td valign="top" align="center">0.769</td>
<td valign="top" align="center">0.714</td>
<td valign="top" align="center">0.786</td>
<td valign="top" align="center">0.733</td>
</tr> <tr>
<td valign="top" align="left">C4</td>
<td valign="top" align="center">0.111</td>
<td valign="top" align="center">0.333</td>
<td valign="top" align="center">0.667</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.444</td>
<td valign="top" align="center">0.571</td>
<td valign="top" align="center">0.400</td>
<td valign="top" align="center">0.667</td>
</tr> <tr>
<td valign="top" align="left">C5</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.788</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.897</td>
<td valign="top" align="center">0.821</td>
<td valign="top" align="center">0.793</td>
<td valign="top" align="center">0.871</td>
<td valign="top" align="center">0.733</td>
</tr> <tr>
<td valign="top" align="left">C6</td>
<td valign="top" align="center">0.824</td>
<td valign="top" align="center">0.7</td>
<td valign="top" align="center">0.882</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.765</td>
<td valign="top" align="center">0.684</td>
<td valign="top" align="center">0.833</td>
<td valign="top" align="center">0.789</td>
</tr> <tr>
<td valign="top" align="left">C7</td>
<td valign="top" align="center">0.793</td>
<td valign="top" align="center">0.742</td>
<td valign="top" align="center">0.862</td>
<td valign="top" align="center">0.833</td>
<td valign="top" align="center">0.724</td>
<td valign="top" align="center">0.808</td>
<td valign="top" align="center">0.742</td>
<td valign="top" align="center">0.639</td>
</tr> <tr>
<td valign="top" align="left">C8</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">C9</td>
<td valign="top" align="center">0.8</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.938</td>
<td valign="top" align="center">0.941</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">Average</td>
<td valign="top" align="center">0.792</td>
<td valign="top" align="center">0.811</td>
<td valign="top" align="center">0.904</td>
<td valign="top" align="center">0.914</td>
<td valign="top" align="center">0.818</td>
<td valign="top" align="center">0.815</td>
<td valign="top" align="center">0.815</td>
<td valign="top" align="center">0.834</td>
</tr> <tr>
<td valign="top" align="left">Kappa</td>
<td valign="top" align="center" colspan="2">0.837</td>
<td valign="top" align="center" colspan="2">0.858</td>
<td valign="top" align="center" colspan="2">0.835</td>
<td valign="top" align="center" colspan="2">0.864</td>
</tr> <tr>
<td valign="top" align="left">OA</td>
<td valign="top" align="center" colspan="2">0.890</td>
<td valign="top" align="center" colspan="2">0.841</td>
<td valign="top" align="center" colspan="2">0.724</td>
<td valign="top" align="center" colspan="2">0.889</td>
</tr></tbody>
</table>
</table-wrap>
<p>When evaluated on the class-balanced dataset (<xref ref-type="table" rid="T7">Table 7</xref>), ResNet101 demonstrated the most consistent performance with the highest minimum values for both recall and precision. Specifically, the lowest recall values were 0.917 for ResNet101 (C6, open-air venues), 0.88 for MobileNet-v2 (C5, low-rise sparse buildings), 0.914 for ResNeXt101_32 &#x000D7; 32d (C6), and 0.917 for Transformer (C6). Similarly, the lowest precision values were 0.904 for ResNet101 (C7, roads), 0.839 for MobileNet-v2 (C7), 0.873 for ResNeXt101_32 &#x000D7; 32d (C7), and 0.941 for Transformer (C7).</p>
<p>Under class-imbalanced conditions (<xref ref-type="table" rid="T8">Table 8</xref>), the Transformer model exhibited superior overall performance, achieving the highest average recall (0.815) and precision (0.834), followed by ResNet101. However, all models struggled significantly with C4 (intersections), which represents the smallest class proportion (0.037, 68/1,858) in the dataset. The C4 recalls were 0.111 (MobileNet-v2), 0.667 (ResNet101), 0.444 (ResNeXt101_32 &#x000D7; 32d), and 0.400 (Transformer). The corresponding precisions were 0.333, 0.75, 0.571, and 0.667, respectively. Notably, the Transformer model demonstrated more balanced performance across categories, with C4&#x00027;s recall-to-average ratio of 0.490 compared to ResNet101&#x00027;s 0.738.</p>
<p>The C4 category exhibited obvious classification bias (<xref ref-type="fig" rid="F8">Figure 8</xref>) due to its severely limited representation. Among conventional CNN architectures, ResNet101 showed the smallest performance gap for C4 relative to other categories. However, the Transformer architecture demonstrated superior robustness to class imbalance, maintaining more consistent performance across all categories. Conversely, C8 (water), with the highest proportion (0.343, 638/1,858) in the class-imbalanced dataset, achieved near-perfect performance (approaching or reaching 100% recall and precision) across all models. The overall accuracy indicators OA and kappa do not adequately represent individual class performance, particularly for minority classes. The performance ranking based on average recall and precision across categories follows: Transformer &#x0003E; ResNet101 &#x0003E; ResNeXt101_32 &#x000D7; 32d &#x0003E; MobileNet-v2, though this hierarchy is not reflected in the overall OA and Kappa metrics.</p>
<fig position="float" id="F8">
<label>Figure 8</label>
<caption><p>Precision and recall of each category in the three models. <bold>(a)</bold> MobileNetv2. <bold>(b)</bold> ResNet101. <bold>(c)</bold> ResNeXt32*32d. <bold>(d)</bold> Transformer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0008.tif">
<alt-text content-type="machine-generated">Four line graphs compare performance metrics across nine classes for different models: MobileNetv2, ResNet101, ResNeXt32x32d, and Transformer. Each graph presents class-balanced recall, class-balanced precision, class-imbalanced recall, and class-imbalanced precision. Trends show variations across classes, with significant dips in metrics particularly noticeable in some classes.</alt-text>
</graphic>
</fig>
<p>C8, with highest proportion in the class-imbalanced dataset (0.343, 638/1,858), has the highest recall and precision in both class-imbalanced and class-balanced results, approaching or reaching 100%.</p>
<p>Data augmentation led to substantial improvements across all architectures. The average OA increased from 0.815 to 0.939, and the average Kappa increased from 0.840 to 0.975. The Transformer model achieved the highest improvements, with OA increasing from 0.889 to 0.951 and kappa from 0.864 to 0.979. Among CNN architectures, ResNeXt101_32 &#x000D7; 32d showed the largest improvement in OA (0.218), while the Transformer model demonstrated the most significant Kappa improvement (0.115). The results reveal distinct architectural characteristics in handling class imbalance. The Transformer model demonstrated superior overall performance and better stability across imbalanced classes, while ResNet101 showed the most consistent performance among CNN architectures. MobileNet-v2, despite being the most lightweight model, exhibited the greatest sensitivity to class imbalance, particularly for minority classes.</p></sec>
<sec>
<label>3.3.2</label>
<title>Visualization analysis</title>
<p>Visualization analysis was conducted based on predicted labels and spatial locations for partial areas (120.3499194&#x000B0;E to 120.3778193&#x000B0;E and 36.0440751&#x000B0;N to 36.0613944&#x000B0;N) as shown in <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F10">10</xref>. Comprehensive comparison demonstrates that models trained on the class-balanced dataset produce results more consistent with satellite imagery, particularly for categories C1 (chaparral), C3 (high-rise sparse buildings), C6 (open-air venues), and C9 (water&#x02013;land junction). Among class-balanced results, ResNet101 achieved optimal classification for C9 and C4, ResNeXt101_32 &#x000D7; 32d excelled in C5 classification, the Transformer model showed superior performance for C2 and C3, while MobileNet-v2 performed best for C2 in specific regions.</p>
<fig position="float" id="F9">
<label>Figure 9</label>
<caption><p>Partial visualization results of the models with the classification legend. <bold>(a)</bold> is VHR image, <bold>(b&#x02013;e)</bold> are Mobilenet-v2, ResNet 101, ResNeXt101_32<sup>&#x0002A;</sup>32d, Transformer results under class-imbalanced respectively, <bold>(f&#x02013;i)</bold> are Mobilenet-v2, ResNet 101, ResNeXt101_32<sup>&#x0002A;</sup>32d, Transformer results under class-balanced respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0009.tif">
<alt-text content-type="machine-generated">A satellite image and eight segmentation maps comparing land use classification models. The first panel (a) shows a very high resolution image of a coastline. Panels (b) to (e) display results from class-imbalanced models: Mobilenet-v2, ResNet101, ResNeXt101_32x32d, and Transformer. Panels (f) to (i) display results from class-balanced models of the same types. The maps use colors to indicate features like water, roads, buildings, and intersections.</alt-text>
</graphic>
</fig>
<fig position="float" id="F10">
<label>Figure 10</label>
<caption><p>Partial visualization results of the models for the detailed block with a classification legend. <bold>(a)</bold> is detailed block image, <bold>(b&#x02013;e)</bold> are Mobilenet-v2, ResNet 101, ResNeXt101_32<sup>&#x0002A;</sup>32d, Transformer results under class-imbalanced respectively, <bold>(f&#x02013;i)</bold> are Mobilenet-v2, ResNet 101, ResNeXt101_32<sup>&#x0002A;</sup>32d, Transformer results under class-balanced respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-07-1613648-g0010.tif">
<alt-text content-type="machine-generated">A set of images shows a detailed aerial view of a block and several classification models applied to the same area. Image (a) is an aerial photograph displaying roads and buildings. Images (b) to (i) show different class-imbalance and class-balance models like Mobilenet-v2, ResNet101, and Transformer applied to the area, using various color-coded segments to classify land features: roads, water, chaparral, intersections, and building density. A legend at the bottom indicates the color representations for each land feature.</alt-text>
</graphic>
</fig>
<p>Detailed examination reveals reduced misclassification of C3 building shadows as C9. In addition, the results suggest that RSSC should be considered a multilabel rather than single-label classification task, as individual scene patches typically contain multiple semantic categories, making multicategory representation more appropriate than single-category assignment.</p>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<label>4</label>
<title>Discussion</title>
<p>The relationship between training sample size and class imbalance sensitivity is fundamental to understanding classifier performance in remote sensing applications. The experimental results demonstrate that categories with fewer training samples experience disproportionately greater impact from class imbalance. Specifically, C4 (intersections), representing the minority class with the lowest proportion (0.037, 68/1,858) in the class-imbalanced dataset, exhibited significant classification deviation across all tested architectures. This phenomenon led to poor convergence during training and notable performance degradation for this category. In contrast, C8 (water), the majority class with the highest proportion (0.343, 638/1,858), achieved near-optimal performance due to its abundant training samples and relatively uniform spectral characteristics. Consequently, all models attained recall and precision values approaching 100% for this category.</p>
<p>Comparative analysis between models trained on class-imbalanced and class-balanced datasets reveals substantial improvements across all evaluation metrics, including overall accuracy, Kappa coefficient, recall, and precision. While the water body (C8) showed minimal performance changes due to its already optimal baseline, the intersections (C4) demonstrated remarkable improvements of 72% and 128% in precision and recall, respectively. This pattern confirms that data augmentation provides the greatest benefits for categories with the most limited training samples, effectively addressing the core challenge of class imbalance.</p>
<p>The relationship between architectural complexity and class imbalance robustness reveals counterintuitive patterns that challenge conventional assumptions. MobileNet-v2, despite its lightweight design with a model size approximately 1/18th that of ResNet101 (8.83 MB vs. 162 MB), achieved competitive performance on balanced datasets with only a 2.6% accuracy gap. However, under class-imbalanced conditions, MobileNet-v2 exhibited significant vulnerability, particularly in minority classes, due to its limited model capacity (<xref ref-type="bibr" rid="B19">Kamilaris and Prenafeta-Bold&#x000FA;, 2018</xref>). More surprisingly, ResNeXt101_32 &#x000D7; 32d, with model complexity approximately 11 times greater than ResNet101 (1,740 MB vs. 162 MB), demonstrated inferior performance on imbalanced datasets despite incorporating advanced architectural features such as grouped convolutions and residual connections. This counterintuitive result suggests that excessive model complexity may lead to overfitting on majority classes while failing to adequately capture minority class representations. The introduction of Transformer (ViT-B/16) provides additional insights, as it achieved superior overall performance and enhanced stability across imbalanced classes, demonstrating that attention-based mechanisms offer inherent advantages for handling class distribution skew.</p>
<p>RSSC inherently involves semantic complexity that extends beyond traditional class imbalance challenges. Individual scene images frequently encompass multiple ground objects with diverse semantic categories, with semantic classification typically determined by the category with the greatest likelihood (<xref ref-type="bibr" rid="B9">Dunne and Campbell, 1997</xref>; <xref ref-type="bibr" rid="B23">Liang et al., 2017</xref>). However, results analysis combined with existing literature reveals that semantic categorization becomes fundamentally ambiguous for many scene images due to the inherent complexity and overlap of ground objects. This observation suggests that multilabel classification frameworks may be more appropriate for remote sensing applications, particularly in heterogeneous urban environments.</p>
<p>Furthermore, data augmentation necessitates category-specific adjustments to mitigate the adverse effects of noisy data (<xref ref-type="bibr" rid="B20">Kim et al., 2003</xref>; <xref ref-type="bibr" rid="B28">Qi et al., 2018</xref>). For example, the cropping scale and position of scene images can influence their semantic category, potentially introducing noisy data through the application of cropping methods. The red-framed region (512<sup>&#x0002A;</sup>512) in <xref ref-type="fig" rid="F10">Figure 10</xref>, classified as an intersection, is segmented into four distinct scene areas (a, b, c, and d) at a 256<sup>&#x0002A;</sup>256 scale. Areas a, b, and c can be categorized as road, while area d represents open-air venues, demonstrating a clear divergence in semantic categories across the two cropping scales. Moreover, a horizontal shift of approximately 128 pixels to the right for area &#x0201C;a&#x0201D; reclassifies it as open-air venues, further illustrating the impact of positional variance on semantic categorization. Additional data augmentation techniques, such as inappropriate rotation angles and excessive HSV disturbance amplitudes, may also contribute to the introduction of noisy data within the dataset.</p>
<p>While conventional augmentation methods prove effective, there remains significant potential for developing specialized techniques that better exploit the unique characteristics of remote sensing data. Future research should explore domain-specific augmentation strategies, including atmospheric variation simulation, multispectral band manipulation, and temporal augmentation that incorporates seasonal and diurnal variations. Such approaches could further enhance performance while maintaining semantic consistency specific to remote sensing imagery characteristics.</p></sec>
<sec sec-type="conclusion" id="s5">
<label>5</label>
<title>Conclusion</title>
<p>Addressing the classification bias caused by class imbalance in RSSC tasks, this study investigated the feasibility of using data augmentation to mitigate class imbalance issues. Four architectures, MobileNet-v2, ResNet101, ResNeXt101_32 &#x000D7; 32d, and Transformer, were selected and fine-tuned using VHR imagery from Shinan district and its surrounding areas. A class-imbalanced high-resolution remote sensing image dataset was constructed, and comprehensive data augmentation methods (mirroring, rotation, cropping, HSV perturbation, and gamma transformation) were used to alleviate the class imbalance problem. The impact of class imbalance on classifier performance was systematically analyzed across all architectures. The results demonstrate that data augmentation represents a realistic and effective approach to mitigating class imbalance problems. Classification bias for minority classes was significantly reduced, with overall performance improvements observed across all classifiers. Among the evaluated models, the Vision Transformer exhibited superior robustness to class imbalance, while ResNet101 demonstrated the most consistent performance among CNN architectures.</p>
<p>Future studies should address several identified limitations. First, inter-class similarity between defined semantic categories artificially increases classification difficulty. Subsequent research should optimize category definitions and sample selection criteria to minimize inter-class confusion while maintaining operational relevance. Second, this study primarily compared performance between severely imbalanced and artificially balanced datasets without exploring intermediate imbalance scenarios. Future investigations should systematically examine varying imbalance ratios, including extreme cases with IR exceeding 100 and long-tailed distributions commonly encountered in large-scale remote sensing applications. In addition, the integration of domain-specific augmentation techniques and multilabel classification frameworks represents promising directions for enhancing model robustness while accommodating the inherent semantic complexity of remote sensing scenes.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>PW: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. XZ: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. YC: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. LZ: Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bria</surname> <given-names>A.</given-names></name> <name><surname>Marrocco</surname> <given-names>C.</given-names></name> <name><surname>Tortorella</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>Addressing class imbalance in deep learning for small lesion detection on medical images</article-title>. <source>Comput. Biol. Med.</source> <volume>120</volume>:<fpage>103735</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compbiomed.2020.103735</pub-id><pub-id pub-id-type="pmid">32250861</pub-id></mixed-citation></ref>
<ref id="B2">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Buda</surname> <given-names>M.</given-names></name> <name><surname>Maki</surname> <given-names>A.</given-names></name> <name><surname>Mazurowski</surname> <given-names>M. A.</given-names></name></person-group> (<year>2018</year>). <article-title>A systematic study of the class imbalance problem in convolutional neural networks</article-title>. <source>Neural Netw.</source> <volume>106</volume>, <fpage>249</fpage>&#x02013;<lpage>259</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2018.07.011</pub-id><pub-id pub-id-type="pmid">30092410</pub-id></mixed-citation></ref>
<ref id="B3">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>G.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name></person-group> (<year>2017</year>). <article-title>Remote sensing image scene classification: benchmark and state of the art</article-title>. <source>Proc. IEEE</source> <volume>105</volume>, <fpage>1865</fpage>&#x02013;<lpage>1883</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JPROC.2017.2675998</pub-id></mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>G.</given-names></name> <name><surname>Sun</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Guo</surname> <given-names>L.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Perturbation-seeking generative adversarial networks: a defense framework for remote sensing image scene classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>60</volume>:<fpage>5605111</fpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2021.3081421</pub-id></mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>G.</given-names></name> <name><surname>Xie</surname> <given-names>X.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name> <name><surname>Guo</surname> <given-names>L.</given-names></name> <name><surname>Xia</surname> <given-names>G. S.</given-names></name></person-group> (<year>2020</year>). <article-title>Remote sensing image scene classification meets deep learning: challenges, methods, benchmarks, and opportunities</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens.</source> <volume>13</volume>, <fpage>3735</fpage>&#x02013;<lpage>3756</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JSTARS.2020.3005403</pub-id></mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>N.</given-names></name></person-group> (<year>2024</year>). <article-title>Masked second-order pooling for few-shot remote-sensing scene classification</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>21</volume>, <fpage>1</fpage>&#x02013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.1109/LGRS.2023.3344840</pub-id></mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Douzas</surname> <given-names>G.</given-names></name> <name><surname>Bacao</surname> <given-names>F.</given-names></name> <name><surname>Last</surname> <given-names>F.</given-names></name></person-group> (<year>2018</year>). <article-title>Improving imbalanced learning through a heuristic oversampling method based on k-means and SMOTE</article-title>. <source>Inf. Sci.</source> <volume>465</volume>, <fpage>1</fpage>&#x02013;<lpage>20</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ins.2018.06.056</pub-id></mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Du</surname> <given-names>B.</given-names></name> <name><surname>Xiong</surname> <given-names>W.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Tao</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Stacked convolutional denoising auto-encoders for feature representation</article-title>. <source>IEEE Trans. Cybern.</source> <volume>47</volume>, <fpage>1017</fpage>&#x02013;<lpage>1027</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TCYB.2016.2536638</pub-id><pub-id pub-id-type="pmid">26992191</pub-id></mixed-citation></ref>
<ref id="B9">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Dunne</surname> <given-names>R.</given-names></name> <name><surname>Campbell</surname> <given-names>N.</given-names></name></person-group> (<year>1997</year>). <article-title>&#x0201C;On the pairing of the softmax activation and cross-entropy penalty functions and the derivation of the softmax activation function,&#x0201D;</article-title> in <source>Proceedings of the 8th Australasian Conference on Neural Networks, Vol. 181</source> (<publisher-loc>Melbourne, VIC</publisher-loc>: <publisher-name>Citeseer</publisher-name>), <fpage>185</fpage>.</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname> <given-names>W.</given-names></name> <name><surname>Huang</surname> <given-names>W.</given-names></name> <name><surname>Bao</surname> <given-names>W.</given-names></name></person-group> (<year>2019</year>). <article-title>Imbalanced hyperspectral image classification with an adaptive ensemble method based on SMOTE and rotation forest with differentiated sampling rates</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>16</volume>, <fpage>1879</fpage>&#x02013;<lpage>1883</lpage>. doi: <pub-id pub-id-type="doi">10.1109/LGRS.2019.2913387</pub-id></mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fern&#x000E1;ndez</surname> <given-names>A.</given-names></name> <name><surname>Garc&#x000ED;a</surname> <given-names>S.</given-names></name> <name><surname>Herrera</surname> <given-names>F.</given-names></name> <name><surname>Chawla</surname> <given-names>N. V.</given-names></name></person-group> (<year>2018</year>). <article-title>SMOTE for learning from imbalanced data: progress and challenges, marking the 15-year anniversary</article-title>. <source>J. Artif. Intell. Res.</source> <volume>61</volume>, <fpage>863</fpage>&#x02013;<lpage>905</lpage>. doi: <pub-id pub-id-type="doi">10.1613/jair.1.11192</pub-id></mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>A survey on deep learning-driven remote sensing image scene understanding: scene classification, scene retrieval and scene-guided object detection</article-title>. <source>Appl. Sci.</source> <volume>9</volume>:<fpage>2110</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app9102110</pub-id></mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Guan</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Feng</surname> <given-names>P.</given-names></name> <name><surname>Shuai</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Meta metric learning for highly imbalanced aerial scene classification,&#x0201D;</article-title> in <source>ICASSP 2020 - 2020 IEEE Int. Conf. Acoust. Speech Signal Process</source> (<publisher-loc>Harbin</publisher-loc>: <publisher-name>College of Computer Science and Technology, Harbin Engineering University; Beijing: China State Key Laboratory of Space-Ground Integrated Information Technology</publisher-name>), <fpage>4042</fpage>&#x02013;<lpage>4046</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICASSP40776.2020.9052900</pub-id></mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harjoseputro</surname> <given-names>Y.</given-names></name> <name><surname>Yuda</surname> <given-names>I. P.</given-names></name> <name><surname>Danukusumo</surname> <given-names>K. P.</given-names></name></person-group> (<year>2020</year>). <article-title>MobileNets: efficient convolutional neural network for identification of protected birds</article-title>. <source>Int. J. Adv. Sci. Eng. Inf. Technol.</source> <volume>10</volume>, <fpage>2290</fpage>&#x02013;<lpage>2296</lpage>. doi: <pub-id pub-id-type="doi">10.18517/ijaseit.10.6.10948</pub-id></mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition kaiming,&#x0201D;</article-title> in <source>Proceedings of the Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>770</fpage>&#x02013;<lpage>778</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Howard</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Chu</surname> <given-names>G.</given-names></name> <name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>B.</given-names></name> <name><surname>Tan</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Searching for MobileNetV3 accuracy vs MADDs vs model size,&#x0201D;</article-title> in <source>2019 IEEE/CVF International Conference on Computer Vision (ICCV)</source> (<publisher-loc>Seoul</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1314</fpage>&#x02013;<lpage>1324</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICCV.2019.00140</pub-id></mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Squeeze-and-excitation networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, <fpage>7132</fpage>&#x02013;<lpage>7141</lpage>.</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Johnson</surname> <given-names>J. M.</given-names></name> <name><surname>Khoshgoftaar</surname> <given-names>T. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Survey on deep learning with class imbalance</article-title>. <source>J. Big Data</source> <volume>6</volume>:<fpage>27</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s40537-019-0192-5</pub-id></mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kamilaris</surname> <given-names>A.</given-names></name> <name><surname>Prenafeta-Bold&#x000FA;</surname> <given-names>F. X.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep learning in agriculture: a survey</article-title>. <source>Comput. Electron. Agric.</source> <volume>147</volume>, <fpage>70</fpage>&#x02013;<lpage>90</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2018.02.016</pub-id></mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>W.</given-names></name> <name><surname>Choi</surname> <given-names>B. J.</given-names></name> <name><surname>Hong</surname> <given-names>E. K.</given-names></name> <name><surname>Kim</surname> <given-names>S. K.</given-names></name> <name><surname>Lee</surname> <given-names>D.</given-names></name></person-group> (<year>2003</year>). <article-title>A taxonomy of dirty data</article-title>. <source>Data Min. Knowl. Discov.</source> <volume>7</volume>, <fpage>81</fpage>&#x02013;<lpage>99</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1021564703268</pub-id></mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leevy</surname> <given-names>J. L.</given-names></name> <name><surname>Khoshgoftaar</surname> <given-names>T. M.</given-names></name> <name><surname>Bauder</surname> <given-names>R. A.</given-names></name> <name><surname>Seliya</surname> <given-names>N.</given-names></name></person-group> (<year>2018</year>). <article-title>A survey on addressing high-class imbalance in big data</article-title>. <source>J. Big Data</source> <volume>5</volume>:<fpage>42</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s40537-018-0151-6</pub-id></mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lewis</surname> <given-names>P. A. W.</given-names></name> <name><surname>McKenzie</surname> <given-names>E.</given-names></name> <name><surname>Hugus</surname> <given-names>D. K.</given-names></name></person-group> (<year>1989</year>). <article-title>Gamma processes</article-title>. <source>Commun. Stat. Stoch. Model.</source> <volume>5</volume>, <fpage>1</fpage>&#x02013;<lpage>30</lpage>. doi: <pub-id pub-id-type="doi">10.1080/15326348908807096</pub-id></mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Lei</surname> <given-names>Z.</given-names></name> <name><surname>Liao</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>S. Z.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Soft-margin softmax for deep classification,&#x0201D;</article-title> in <source>Lect. Notes Comput. Sci. (Including Subser. Lect. Notes Artif. Intell. Lect. Notes Bioinformatics)</source>, vol. 10635 (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>413</fpage>&#x02013;<lpage>421</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-3-319-70096-0_43</pub-id></mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Tsoumakas</surname> <given-names>G.</given-names></name></person-group> (<year>2020</year>). <article-title>Dealing with class imbalance in classifier chains via random undersampling</article-title>. <source>Knowl. -Based Syst.</source> <volume>192</volume>:<fpage>105292</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.knosys.2019.105292</pub-id></mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <name><surname>Yao</surname> <given-names>R.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Zheng</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Siamese convolutional neural networks for remote sensing scene classification</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>16</volume>, <fpage>1200</fpage>&#x02013;<lpage>1204</lpage>. doi: <pub-id pub-id-type="doi">10.1109/LGRS.2019.2894399</pub-id></mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Luque</surname> <given-names>A.</given-names></name> <name><surname>Carrasco</surname> <given-names>A.</given-names></name> <name><surname>Mart&#x000ED;n</surname> <given-names>A.</given-names></name> <name><surname>de las Heras</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>The impact of class imbalance in classification performance metrics based on the binary confusion matrix</article-title>. <source>Pattern Recognit.</source> <volume>91</volume>, <fpage>216</fpage>&#x02013;<lpage>231</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.patcog.2019.02.023</pub-id></mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>D.</given-names></name> <name><surname>Tang</surname> <given-names>P.</given-names></name> <name><surname>Zhao</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>SiftingGAN: generating and sifting labeled samples to improve the remote sensing image scene classification baseline <italic>in vitro</italic></article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>16</volume>, <fpage>1046</fpage>&#x02013;<lpage>1050</lpage>. doi: <pub-id pub-id-type="doi">10.1109/LGRS.2018.2890413</pub-id></mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Gao</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Impacts of dirty data: and experimental evaluation</article-title>. <source>arXiv preprint</source> arXiv:1803.06071.</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Full convolutional neural network based on multi-scale feature fusion for the class imbalance remote sensing image classification</article-title>. <source>Remote Sens.</source> <volume>12</volume>, <fpage>1</fpage>&#x02013;<lpage>21</lpage>. doi: <pub-id pub-id-type="doi">10.3390/rs12213547</pub-id></mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Sandler</surname> <given-names>M.</given-names></name> <name><surname>Howard</surname> <given-names>A.</given-names></name> <name><surname>Zhu</surname> <given-names>M.</given-names></name> <name><surname>Zhmoginov</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;MobileNetV2: inverted residuals and linear bottlenecks,&#x0201D;</article-title> in <source>018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4510</fpage>&#x02013;<lpage>4520</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR.2018.00474</pub-id><pub-id pub-id-type="pmid">39300076</pub-id></mixed-citation></ref>
<ref id="B31">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>Z.</given-names></name> <name><surname>Feng</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>E.</given-names></name> <name><surname>He</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>Normalised gamma transformation-based contrast-limited adaptive histogram equalisation with colour correction for sand-dust image enhancement</article-title>. <source>IET Image Process.</source> <volume>14</volume>, <fpage>747</fpage>&#x02013;<lpage>756</lpage>. doi: <pub-id pub-id-type="doi">10.1049/iet-ipr.2019.0992</pub-id></mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Sermanet</surname> <given-names>P.</given-names></name> <name><surname>Reed</surname> <given-names>S.</given-names></name> <name><surname>Anguelov</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2015</year>). &#x0201C;Going deeper with convolutions&#x0201D;, in <italic>In Proceedings of the Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</italic> (Boston, MA: IEEE), <fpage>1</fpage>&#x02013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298594</pub-id></mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>X.</given-names></name> <name><surname>Ma</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Ma</surname> <given-names>J.</given-names></name> <name><surname>Jiao</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>Attention consistent network for remote sensing scene classification</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens.</source> <volume>14</volume>, <fpage>2030</fpage>&#x02013;<lpage>2045</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JSTARS.2021.3051569</pub-id></mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thabtah</surname> <given-names>F.</given-names></name> <name><surname>Hammoud</surname> <given-names>S.</given-names></name> <name><surname>Kamalov</surname> <given-names>F.</given-names></name> <name><surname>Gonsalves</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Data imbalance in classification: experimental evaluation</article-title>. <source>Inf. Sci.</source> <volume>513</volume>, <fpage>429</fpage>&#x02013;<lpage>441</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ins.2019.11.004</pub-id></mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Doll</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Aggregated residual transformations for deep,&#x0201D;</article-title> in <source>2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1492</fpage>&#x02013;<lpage>1500</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR.2017.634</pub-id></mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Yessou</surname> <given-names>H.</given-names></name> <name><surname>Sumbul</surname> <given-names>G.</given-names></name> <name><surname>Demir</surname> <given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;A comparative study of deep learning loss functions for multi-label remote sensing image classification,&#x0201D;</article-title> in <source>IGARSS 2020 - 2020 IEEE International Geoscience and Remote Sensing Symposium</source> (<publisher-loc>Waikoloa, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1349</fpage>&#x02013;<lpage>1352</lpage>. doi: <pub-id pub-id-type="doi">10.1109/IGARSS39084.2020.9323583</pub-id></mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yosinski</surname> <given-names>J.</given-names></name> <name><surname>Clune</surname> <given-names>J.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Lipson</surname> <given-names>H.</given-names></name></person-group> (<year>2014</year>). <article-title>How transferable are features in deep neural networks?</article-title> <source>Adv. Neural Inf. Process. Syst.</source> <volume>4</volume>, <fpage>3320</fpage>&#x02013;<lpage>3328</lpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1411.1792</pub-id></mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>Attention GANs: unsupervised deep feature learning for aerial scene classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>58</volume>, <fpage>519</fpage>&#x02013;<lpage>531</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2019.2937830</pub-id></mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>J.</given-names></name> <name><surname>Kong</surname> <given-names>L.</given-names></name> <name><surname>Lv</surname> <given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>An overview of deep neural networks for few-shot learning</article-title>. <source>Big Data Min. Anal.</source> <volume>8</volume>, <fpage>145</fpage>&#x02013;<lpage>188</lpage>. doi: <pub-id pub-id-type="doi">10.26599/BDMA.2024.9020049</pub-id></mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Xiong</surname> <given-names>S.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name> <name><surname>Zhu</surname> <given-names>X. X.</given-names></name> <name><surname>Mou</surname> <given-names>L.</given-names></name></person-group> (<year>2024a</year>). <article-title>Co-enhanced global-part integration for remote-sensing scene classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>62</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2024.3367877</pub-id></mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Gong</surname> <given-names>M.</given-names></name> <name><surname>Qin</surname> <given-names>A. K.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Hu</surname> <given-names>Z.</given-names></name> <name><surname>Gao</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2024b</year>). <article-title>Gradient-guided multiscale focal attention network for remote sensing scene classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>62</volume>, <fpage>1</fpage>&#x02013;<lpage>18</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2024.3424489</pub-id></mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/132078/overview">Marcello Pelillo</ext-link>, Ca&#x00027; Foscari University of Venice, Italy</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1392526/overview">Zhen Shen</ext-link>, Nanyang Institute of Technology, China; <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1717829/overview">Nanqing Liu</ext-link>, Southwest Jiaotong University, China; <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2961357/overview">Yue Zhao</ext-link>, Xidian University, China</p>
</fn>
</fn-group>
</back>
</article>