<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Earth Sci.</journal-id>
<journal-title>Frontiers in Earth Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Earth Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-6463</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1268628</article-id>
<article-id pub-id-type="doi">10.3389/feart.2023.1268628</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Earth Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Extraction of building from remote sensing imagery base on multi-attention L-CAFSFM and MFFM</article-title>
<alt-title alt-title-type="left-running-head">Jin et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/feart.2023.1268628">10.3389/feart.2023.1268628</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Jin</surname>
<given-names>Huazhong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2544615/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fu</surname>
<given-names>Wenjun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2390165/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nie</surname>
<given-names>Chenhui</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yuan</surname>
<given-names>Fuxiang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2392974/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chang</surname>
<given-names>Xueli</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1976263/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Computer Science</institution>, <institution>Hubei University of Technology</institution>, <addr-line>Wuhan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Zhejiang Academy of Surveying and Mapping</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>College of Management</institution>, <institution>Jianghan Art Vocational College</institution>, <addr-line>Qianjiang</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1140491/overview">Bahareh Kalantar</ext-link>, Riken, Japan</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1423762/overview">Vahideh Saeidi</ext-link>, Putra Malaysia University, Malaysia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1045747/overview">Mohammed Oludare Idrees</ext-link>, University of Abuja, Nigeria</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Xueli Chang, <email>chang99@hbut.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1268628</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>10</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Jin, Fu, Nie, Yuan and Chang.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Jin, Fu, Nie, Yuan and Chang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Building extraction from high-resolution remote sensing images is widely used in urban planning, land resource management, and other fields. However, the significant differences between categories in high-resolution images and the impact of imaging, such as atmospheric interference and lighting changes, make it difficult for high-resolution images to identify buildings. Therefore, detecting buildings from high-resolution remote sensing images is still challenging. In order to improve the accuracy of building extraction in high-resolution images, this paper proposes a building extraction method combining a bidirectional feature pyramid, location-channel attention feature serial fusion module (L-CAFSFM), and meticulous feature fusion module (MFFM). Firstly, richer and finer building features are extracted using the ResNeXt101 network and deformable convolution. L-CAFSFM combines feature maps from two adjacent levels and iteratively calculates them from high to low level, and from low to high level, to enhance the model&#x2019;s feature extraction ability at different scales and levels. Then, MFFM fuses the outputs from the two directions to obtain building features with different orientations and semantics. Finally, a dense conditional random field (Dense CRF) improves the correlation between pixels in the output map. Our method&#x2019;s precision, F-score, Recall, and IoU(Intersection over Union) on WHU Building datasets are 95.17%&#x3001;94.83%&#x3001;94.51% and 90.18%. Experimental results demonstrate that our proposed method has a more accurate effect in extracting building features from high-resolution image.</p>
</abstract>
<kwd-group>
<kwd>remote sensing image</kwd>
<kwd>building detection</kwd>
<kwd>building extraction</kwd>
<kwd>location-channel attention feature serial fusion module (L-CAFSFM)</kwd>
<kwd>meticulous feature fusion module (MFFM)</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Environmental Informatics and Remote Sensing</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>With the rapid development of sub-meter-level high-resolution earth observation satellites, it has become possible to obtain high-resolution images of the surface over a large area. Buildings are one of the most common and essential elements of the earth&#x2019;s surface (<xref ref-type="bibr" rid="B5">Cai, et al., 2021</xref>; <xref ref-type="bibr" rid="B29">Sheikh, et al., 2022</xref>; <xref ref-type="bibr" rid="B43">Yuan and Mohd Shafri, 2022</xref>; <xref ref-type="bibr" rid="B44">Zhang, et al., 2022</xref>). Building detection from remote sensing images has been widely used in urban development planning, land development and utilization, post-disaster damage assessment, and other fields. Factors such as the size, shape, texture difference, cloud occlusion, surface material reflection, and shadow of buildings in remote sensing images can reduce the accuracy of building detection. Improving the accuracy of building detection has essential application value and practical significance for urban 3D modeling, map updating, disaster assessment, etc (<xref ref-type="bibr" rid="B2">Bauchet, et al., 2021</xref>; <xref ref-type="bibr" rid="B6">Chen and Sun, 2022</xref>; <xref ref-type="bibr" rid="B13">Fang, et al., 2022</xref>; <xref ref-type="bibr" rid="B16">Hou, et al., 2022</xref>; <xref ref-type="bibr" rid="B40">Yang, et al., 2022</xref>).</p>
<p>Traditional methods of extracting buildings from remote sensing images mainly rely on manually extracted features, such as brightness, texture, shape, and prior knowledge. (<xref ref-type="bibr" rid="B22">Lin and Zhang, 2017</xref>) proposed an object-based morphological building index (OBMBI) by comprehensively using image segmentation and graph-based mathematical morphology top-hat reconstruction technology, using image segmentation to obtain objects and establish topological relationship diagrams between objects. The feature function of the graph is created using the brightness value feature of the object, and a bidirectional mapping relationship between the pixels, objects, and graph nodes is established. Morphological building index maps are generated using top-hat reconstruction techniques. But this method is only suitable for remote sensing images with a resolution better than 1&#xa0;m. (<xref ref-type="bibr" rid="B25">Ma, et al., 2019</xref>) proposed a new morphological attribute building index (MABI), which establishes morphological attribute filters (AFs) with the building features of the input images (Such as high local contrast, internal homogeneity, shape, and size) and is used for image segmentation to obtain building regions with high homogeneity. However, this method&#x2019;s segmentation standard deviation threshold must be manually set. (<xref ref-type="bibr" rid="B45">Zhang, et al., 2019</xref>) used a novel rotation uniform invariant local binary pattern algorithm to obtain low-density feature maps and use mean shifts to extract building edges. This method can accurately segment the boundary of simple buildings, but the detection effect could be better when there are many buildings and complex scenes. (<xref ref-type="bibr" rid="B32">Wang, et al., 2019</xref>) proposed the Adaptive Morphological Attribute Profile under Object Boundary Constraint (AMAP-OBC) method under the constraints of building boundaries and combined with Morphological Attribute Profiles (MAPs). In the preprocessing step, candidate object sets are extracted through MAPs. Secondly, the candidate object set is processed by AMAP-OBC to obtain the initial building set. Finally, the building sets are segmented using an adaptive threshold to obtain final building extraction results. However, the morphological property profile of this method is challenging to obtain in advance.</p>
<p>Because traditional building extraction methods for high-resolution remote sensing images often rely on low-level features, and the means of describing and representing building features are single and specific, it is challenging to extract the features of different types of buildings (<xref ref-type="bibr" rid="B3">Borba, et al., 2021</xref>; <xref ref-type="bibr" rid="B36">Xiao, et al., 2022</xref>). In fact, traditional building extraction methods are not universal and cannot meet the needs of most scenes. Assuming that the high-level semantic features of high-resolution remote sensing images are utilized, along with their low-level features (<xref ref-type="bibr" rid="B38">Xie, et al., 2020</xref>; <xref ref-type="bibr" rid="B31">Tian, et al., 2022</xref>), such as shape, texture, brightness, and contour, different levels of features are fused, which can improve the accuracy of building extraction.</p>
<p>In recent years, deep learning technology has led the continuous in-depth application and expansion of artificial intelligence in different industries and fields, especially in computer vision (<xref ref-type="bibr" rid="B7">Chen, et al., 2022a</xref>; <xref ref-type="bibr" rid="B30">Shi, et al., 2022</xref>; <xref ref-type="bibr" rid="B34">Wei, et al., 2022</xref>). The reason is that deep learning can manipulate data or symbols to form different levels of features to recognize patterns, model approximate functions in the data or symbols, and interpret and understand what people see. Deep learning, as a learning mechanism, writes rules for specific patterns or defines symbols for fuzzy concepts through self-supervised learning of large amounts of data. Therefore, image-building extraction can define and describe a pattern or concept from many images (<xref ref-type="bibr" rid="B1">Abdollahi, et al., 2020</xref>; <xref ref-type="bibr" rid="B20">Li, et al., 2020a</xref>; <xref ref-type="bibr" rid="B28">Saini, et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Chen, et al., 2022b</xref>; <xref ref-type="bibr" rid="B39">Yan, et al., 2022</xref>; <xref ref-type="bibr" rid="B41">You, et al., 2022</xref>).</p>
<p>Some researchers have used deep learning methods to segment buildings in high-resolution remote sensing images, significantly improving detection accuracy. <xref ref-type="bibr" rid="B42">Yu, et al. (2021)</xref> proposed the Capsule Feature Pyramid Network (CapFPN), which utilizes the characteristics of the feature pyramid and fuses the features of the capsule network at different levels. CapFPN can extract features with high resolution and intrinsic solid semantics, effectively improving the extraction accuracy of pixel-level buildings. (<xref ref-type="bibr" rid="B48">Zhu, et al., 2021</xref>) used Multi Attending Path Neural Network (MAP-Net) to learn multi-scale features in feature space. They use an attention module to adaptively compress the features of each channel, which is used to fuse multi-scale features. Then, global dependency can be captured using a pyramid spatial pooling module to optimize discontinuous buildings. (<xref ref-type="bibr" rid="B47">Zhu, et al., 2018</xref>) used a Bidirectional Feature Pyramid Network (BFPN) to fuse feature maps of different scales and enhance the feature encoding ability. The above methods use the feature pyramid structure to extract and fuse multi-scale features. However, the feature information extracted by a single feature pyramid network must be more prosperous, and the model&#x2019;s ability to perceive features needs to be improved.</p>
<p>The attention mechanism is widely used in the field of image processing, inspired by the research on human attention. In image analysis, the attention mechanism can focus on important feature information with high weight and ignore irrelevant information with low weight. (<xref ref-type="bibr" rid="B14">Guo, et al., 2020</xref>) proposed a U-Net building extraction method with attention modules and multiple losses. It can improve the model&#x2019;s sensitivity through the attention module and suppress the background influence of irrelevant feature areas. However, as a fully supervised method, it relies on many manual label samples. (<xref ref-type="bibr" rid="B12">Das and Chand, 2021</xref>) proposed Attention Building Net (ABNet), which utilizes a convolutional attention module with a channel and spatial attention mechanism to focus on essential features selectively. Building boundaries can be accurately extracted because it improves the overall feature representation. However, this method needs to pay attention to the correlation between features of different levels and scales, resulting in poor detection in complex scenes. (<xref ref-type="bibr" rid="B4">Cai and Chen, 2021</xref>) designed a downsampling module combining separable convolution and channel attention to extract features from the input graph. However, single-channel attention only pays attention to the channel information of features and needs help to obtain good feature space information. (<xref ref-type="bibr" rid="B21">Li, et al., 2020b</xref>) used a convolutional neural network to extract pixel-level building shadows and used conditional random fields (CRF) as post-processing optimization experimental results, which achieved good results. However, CRF needs to use the correlation between pixels; there is still room for improvement.</p>
<p>The above scholars have provided different methods to improve the model network and attention mechanism. However, some methods still need to be improved, such as insufficient extraction of features from a single model, insufficient attentional fusion, and failure to consider correlations between features in neighboring hierarchies. In response to these issues, this paper designs the location-channel attention feature serial fusion module (L-CAFSFM) and the meticulous feature fusion module (MFFM) in the bidirectional feature pyramid network. First, based on the ResNeXt101 network, combined with deformable convolution, a group of feature maps with different levels and resolutions is generated. Then, the L-CAFSFM is used to iteratively calculate the two adjacent feature maps from low level to high level and from high level to low level. MFFM is used to fuse the output of two directions. Finally, Dense Conditional Random Field (Dense CRF) is applied to optimize the results and output the prediction image.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methodology</title>
<p>In this section, we will elaborate on our method. First, the overall network structure diagram is introduced. Then, the Location-Channel Attention Feature Serial Fusion module (L-CAFSFM), Meticulous Feature Fusion module (MFFM), and loss function are introduced in detail.</p>
<sec id="s2-1">
<title>2.1 Model overview</title>
<p>Traditional neural network models want to improve accuracy by deepening or widening the network. However, with the increase of super parameters (such as the number of channels and convolution size), the difficulty of network design and computational expense will increase. ResNext deep neural network can improve the accuracy without increasing the complexity of the parameters and also reduce the number of super parameters. Based on VGG/ResNets&#x2019; duplicate layer strategy and split transform merge strategy, the ResNext101 network proposes an aggregate transformations method, which uses a parallel stack of blocks with the same topology structure to replace the original ResNets&#x2019; three-layer convolutional block. The model&#x2019;s accuracy is improved without significantly increasing the number of parameters. At the same time, because of the same topology, the super parameters are reduced accordingly.</p>
<p>Traditional convolution kernels&#x2019; size is usually fixed (e.g., 3 &#xd7; 3, 5 &#xd7; 5, 7 &#xd7; 7). They have poor adaptability to changes in unknown objects and weak generalization ability. Deformable convolution introduces a learnable offset in the receptive field so that the receptive field becomes a polygon, which is no longer limited to a square, and can extract more accurate features at different levels. Therefore, we use the ResNeXt101 network (<xref ref-type="bibr" rid="B37">Xie, et al., 2017</xref>)combined with deformable convolution (<xref ref-type="bibr" rid="B11">Dai, et al., 2017</xref>) to extract feature maps of different scales and levels of the input image.</p>
<p>The network structure diagram proposed in this paper is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The overall structure of the network. Deform Conv denotes deformable convolution, Concat denotes concatenating, and Up denotes upsampling.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g001.tif"/>
</fig>
<p>High-level features contain rich semantic information about buildings, and low-level feature maps have fine local detail features of buildings. Inspired by the article (<xref ref-type="bibr" rid="B48">Zhu, et al., 2021</xref>), we concatenate adjacent high-level features with low-level features and input them into the L-CAFSFM for computation. Two different directions are used to calculate semantic information iteratively: one is from high-level to low-level, and the other is the opposite, to obtain multi-scale information in different directions and levels. Then, MFFM fuses the outputs from both directions. Finally, dense conditional random field (Dense CRF) improves the correlation of each pixel in the output image to obtain a building prediction map.</p>
</sec>
<sec id="s2-2">
<title>2.2 Location-channel attention feature serial fusion</title>
<p>Attention mechanisms can focus on the more critical information of the current task in a large amount of information, reduce attention to other information, and even filter out irrelevant information to improve the efficiency and accuracy of task processing. The attention mechanism is widely applied in computer vision fields such as image segmentation and classification and is roughly divided into three categories: spatial, channel, and spatial-channel hybrid attention. SE (Squeeze-and-Excitation) attention (<xref ref-type="bibr" rid="B18">Jie, et al., 2017</xref>) is typical channel attention, which only considers the internal channel information and ignores the importance of location information. BAM(Bottleneck Attention Module) (<xref ref-type="bibr" rid="B26">Park, et al., 2018</xref>) and CBAM(Convolutional Block Attention Module) (<xref ref-type="bibr" rid="B35">Woo, et al., 2018</xref>) try to introduce location information by global pooling on channels, but they can only capture local information instead of long-range dependent information. The self-attention is an improvement of the attention mechanism, which reduces the dependence on external information and is better at capturing the internal correlation of features. However, when using the self-attention mechanism to encode the information about the current position, the model will excessively focus on its own position.</p>
<p>Given the above attention mechanism problems, we introduce coordinate attention and Swin Transformer Block to build the Location-Channel Attention Feature Serial Fusion module (L-CAFSFM). Coordinate attention can improve the ability to obtain location information and channel information and reduce the loss of channel information and location information caused by downsampling operations (<xref ref-type="bibr" rid="B15">Hou, et al., 2021</xref>). Swin Transformer Block can capture the internal correlation of location and channel information and improve the network&#x2019;s sensitivity to information (<xref ref-type="bibr" rid="B24">Liu, et al., 2021</xref>). The L-CAFSFM is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The structure of L-CAFSFM.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g002.tif"/>
</fig>
<p>L-CAFSFM can be divided into feature location enhancement and feature channel information optimization.</p>
<sec id="s2-2-1">
<title>2.2.1 Feature location enhancement</title>
<p>In the feature location enhancement step, the Coordinate Attention Block calculates the input feature map. Coordinate attention captures feature details across channels and includes feature orientation and location information. It enables the model to locate and identify target regions more accurately and enhances the model&#x2019;s feature expression capabilities. Coordinate attention can take an arbitrary <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> as input and transform it into <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, which is output with the same size and the same channel as <italic>X</italic>.</p>
<p>Coordinate attention encodes channel relationships and long-term dependencies with precise location information and can be divided into coordinate information embedding and attention generation. The structure of coordinate attention is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The structure of coordinate attention.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g003.tif"/>
</fig>
<p>In the coordinate information embedding module, global average pooling is performed respectively on the horizontal and vertical directions of the input feature map so that the attention module can capture the interaction information in different directions and different spaces. Specifically, for the input feature <italic>X</italic>, the pooling kernels of size (1, <italic>W</italic>) and (<italic>H</italic>, 1) are used to encode along the vertical and horizontal directions, respectively, so the output of the c-<italic>th</italic> channel at height <italic>h</italic> is:<disp-formula id="e1">
<mml:math id="m3">
<mml:mrow>
<mml:msubsup>
<mml:mi>Y</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>the output of the c-<italic>th</italic> channel at height <italic>h</italic> is:<disp-formula id="e2">
<mml:math id="m4">
<mml:mrow>
<mml:msubsup>
<mml:mi>Y</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>By aggregating features along two spatial directions, a pair of direction-aware feature maps can be obtained, enabling the attention module to capture long-term dependencies along one spatial direction and preserve precise location information along the other. It helps the network to more accurately locate the target of interest.</p>
<p>In the coordinate attention generation module, a better global receptive field can be obtained after the above transformation, and the precise location information of the feature can be encoded. In order to capture the information between channels simultaneously, the two outputs of the module in the previous step are concatenated and computed through the convolution transformation function.<disp-formula id="e3">
<mml:math id="m5">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>Y</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>Y</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>Concat</italic> (&#xb7;) is a concatenation operation along the horizontal and vertical directions, and <italic>Nl</italic> (&#xb7;) is a nonlinear activation function. <italic>F</italic> is the intermediate feature vector that encodes the spatial information in the horizontal and vertical directions. Then it is decomposed into two separate vectors <inline-formula id="inf3">
<mml:math id="m6">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf4">
<mml:math id="m7">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> along the horizontal and vertical directions, which are activated by convolution transformation and Sigmoid function, respectively. The results are:<disp-formula id="e4">
<mml:math id="m8">
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m9">
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf5">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#xb7;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf6">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#xb7;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represent convolution operations on <inline-formula id="inf7">
<mml:math id="m12">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m13">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The output <italic>T</italic> of the last Coordinate Attention Block is:<disp-formula id="e6">
<mml:math id="m14">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf9">
<mml:math id="m15">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#xb7;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the Re-Weight operation, that is, restore <inline-formula id="inf10">
<mml:math id="m16">
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf11">
<mml:math id="m17">
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to <italic>C</italic>&#xd7;<italic>H</italic>&#xd7;<italic>W</italic> size. The output <inline-formula id="inf12">
<mml:math id="m18">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> has the same size and dimension as the input <italic>X</italic>. We compress <italic>T</italic> into <inline-formula id="inf13">
<mml:math id="m19">
<mml:mrow>
<mml:msup>
<mml:mi>T</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. <italic>T&#x2032;</italic> is a row vector with dimension <italic>C</italic> and size [1, (<inline-formula id="inf14">
<mml:math id="m20">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>)]. <italic>T&#x2032;</italic> is used as the output of this stage.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Feature channel information optimization</title>
<p>In optimizing feature channel information, we choose Swin Transformer Block in the article (<xref ref-type="bibr" rid="B24">Liu, et al., 2021</xref>) to calculate the output of the previous step. Swin Transformer Block consists of a self-attention based on a sliding window and a self-attention without a sliding window. Multi-head attention is added to self-attention, which can extract feature information from multiple dimensions. Constraining attention computation within a window through multiple windows and using Shifted Window to link multiple windows make it easier to capture fine local features. Therefore, this paper designs a feature optimization module based on Swin Transformer Block, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Structure of feature optimization module.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g004.tif"/>
</fig>
<p>The main body of the module consists of two Swin Transformer Blocks. W-MSA (Window-based Multi-head Self Attention) is a multi-head self-attention without a sliding window. Different from the global self-attention calculation, W-MSA can calculate self-attention. It reduces the amount of computation without causing a lot of memory consumption.</p>
<p>Because self-attention is calculated in multiple windows, the information between windows cannot interact, and the effect of global modeling cannot be achieved. To solve this problem, the article (<xref ref-type="bibr" rid="B24">Liu, et al., 2021</xref>) proposes the SW-SAM module. SW-SAM (Shifted Window-based Multi-head Self Attention) is a multi-head self-attention with a sliding window. By sliding the window in the feature map, the information of different windows is collected, and the communication between the windows is realized to establish a global model. MLP is a multilayer perceptron that performs nonlinear classification of features. This module calculates the output of the previous step: as the input dimension <inline-formula id="inf15">
<mml:math id="m21">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, after two Swin Transformer Block calculations, the output <inline-formula id="inf16">
<mml:math id="m22">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the same as the input dimension, and <inline-formula id="inf17">
<mml:math id="m23">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a row vector with <italic>C</italic> channels and size [1, (<italic>H</italic> <inline-formula id="inf18">
<mml:math id="m24">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <italic>W</italic>)]. Finally, <inline-formula id="inf19">
<mml:math id="m25">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is restored to a feature map <inline-formula id="inf20">
<mml:math id="m26">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> according to the number of channels <italic>C</italic> and the size of [<italic>C, H, W</italic>] as the output of L-CAFSFM.</p>
<p>As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, the characteristic diagram is not processed by the L-CAFSFM and is calculated by the L-CAFSFM.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> is the input image. <xref ref-type="fig" rid="F2">Figure 2</xref> shows the feature map the L-CAFSFM still needs to calculate. The distinction between the building area and the trees and roads in the red circle needs to be more apparent, and the overall brightness value of the feature map is close to the building area. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the feature map calculated by L-CAFSFM. In <xref ref-type="fig" rid="F3">Figure 3</xref>, the building area&#x2019;s features differ from other features, suppressing the other features. It shows that after L-CAFSFM calculation, the model can extract more accurate and rich building features.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Meticulous feature fusion module</title>
<p>After the iterative calculation of the L-CAFSFM by the bidirectional feature pyramid network, fusing the outputs of two opposite paths is necessary. Inspired by the article (<xref ref-type="bibr" rid="B47">Zhu, et al., 2018</xref>), we propose the Meticulous Feature Fusion module (MFFM). The module structure is shown in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The structure of MFFM.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g006.tif"/>
</fig>
<p>For the outputs <inline-formula id="inf21">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf22">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> of the two opposite paths, use a <inline-formula id="inf23">
<mml:math id="m29">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution operation to connect the two outputs to obtain the meticulous feature <inline-formula id="inf24">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="e7">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <italic>Concat</italic> (&#xb7;) represents the connection operation, and <inline-formula id="inf25">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#xb7;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula id="inf26">
<mml:math id="m33">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution operation.</p>
<p>After connecting <inline-formula id="inf27">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf28">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the channel of the feature map is 2<italic>C</italic>, and the size is [<italic>H, W</italic>]. Input it into this paper&#x2019;s improved SE Attention (Fusion-SE), and then use the Sigmoid operation. The output is a meticulous feature <inline-formula id="inf29">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with channel two and size [<italic>H, W</italic>]. It is:<disp-formula id="e8">
<mml:math id="m37">
<mml:mrow>
<mml:mfenced open="" close="}" separators="|">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <italic>&#x3d5;</italic>(&#xb7;) represents the sigmoid operation, and <inline-formula id="inf30">
<mml:math id="m38">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#xb7;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the Fusion-SE attention. It is the Fusion-SE attention module diagram, as shown in <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The structure of fusion-SE attention.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g007.tif"/>
</fig>
<p>In <xref ref-type="fig" rid="F7">Figure 7</xref>, GAP stands for global average pooling. After inputting <inline-formula id="inf31">
<mml:math id="m39">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> into the SE attention module, the output of <inline-formula id="inf32">
<mml:math id="m40">
<mml:mrow>
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is consistent with the input dimension and size. After convolution transformation, batch normalization and linear activation, the output is <inline-formula id="inf33">
<mml:math id="m41">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The fusion-SE attention module reduces the dimension <italic>C</italic> of the input <italic>K</italic> to 2, which is convenient for calculation with the meticulous feature <inline-formula id="inf34">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the previous step.</p>
<p>The final output of building a prediction map is:<disp-formula id="e9">
<mml:math id="m43">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>Dense CRF optimizes the prediction map to improve the correlation between different pixels in the prediction map. It can be seen from <xref ref-type="fig" rid="F7">Figure 7</xref> that after Dense CRF optimization of the prediction map, the building edge details have been further optimized, which is closer to the label map. Thus, the final building detection map of our method is obtained, as shown in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> is the label image, <xref ref-type="fig" rid="F2">Figure 2</xref> shows the extraction results without Dense CRF optimization, and <xref ref-type="fig" rid="F3">Figure 3</xref> shows the extraction results with Dense CRF optimization. Comparing objects of the red circle box in <xref ref-type="fig" rid="F2">Figure 2</xref> and <xref ref-type="fig" rid="F3">Figure 3</xref>, it is clear that the extracted edges of the buildings are clearer and more regular after the Dense CRF optimization of the predicted maps.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g008.tif"/>
</fig>
</sec>
</sec>
<sec id="s3">
<title>3 Experiments and results</title>
<sec id="s3-1">
<title>3.1 Dataset and experimental settings</title>
<p>In order to evaluate the method proposed in this paper, we selected the public building dataset (WHU) of Wuhan University. WHU is an aerial image dataset. This dataset consists of aerial images obtained in April 2012 and covers an area of 20.5&#xa0;km<sup>2</sup> 12,796 buildings. The aerial image data comes from the New Zealand Land Information Service website, with a ground resolution of 0.3&#xa0;m after low sampling, selected from approximately 22,000 buildings in Christchurch. This dataset contains 8,188 remote sensing images and has a resolution of 512 &#xd7; 512 pixels, covering residential, factory, urban, rural, etc. buildings in the area. The training set size is 4,736 images, the validation set size is 1,036 images, and the test set size is 2,416.</p>
<p>The training platform used in this paper has a 24G GPU and an 8-core CPU. Train the model using the training set from the WHU dataset. The training batch size is 4, the number of training times is 15w, and the initial learning rate is 0.005. The entire training network is optimized using Stochastic Gradient Descent (SGD) with a weight decay 0.0005. We use a deep supervision method (<xref ref-type="bibr" rid="B19">Lee, et al., 2014</xref>) to set a branch classifier on the output of each L-CAFSFM to supervise the quality of the output, thus facilitating the dissemination of helpful information. The loss of the L-CAFSFM in each direction in the bidirectional feature pyramid network is calculated, and the total loss is the sum of the losses in the two directions.</p>
</sec>
<sec id="s3-2">
<title>3.2 Comparative test</title>
<p>We use the test set in the WHU dataset to test our model. The test set has 2,416 remote sensing images and has a resolution of 512 &#xd7; 512 pixels. The experiment in this paper is compared with that in the original paper (<xref ref-type="bibr" rid="B24">Liu, et al., 2021</xref>), and some results are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Comparison of the accuracy of our method and BDRAR in extracting building shape. The parts circled in red in the picture show significant differences in comparison. In the above figure, the first column is the input map, the second column is the label map, the third column is the experimental result map of the BDRAR model (<xref ref-type="bibr" rid="B47">Zhu, et al., 2018</xref>), and the fourth column is the experimental result map of this paper.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g009.tif"/>
</fig>
<p>As we can see from the first and second rows of <xref ref-type="fig" rid="F9">Figure 9</xref>. The results of the BDRAR model showed some missing inspections, cavities, or missing corners, resulting in an incomplete building shape. The shapes of the buildings detected in this paper are relatively regular and complete. In the third row, the BDRAR model identifies multiple buildings with close distances into one, while this paper can clearly display the boundaries of multiple buildings. It can be seen from the fourth and fifth rows that when BDRAR model distinguishes buildings and open spaces in front of doors, it mistakenly detects open spaces as buildings, and some buildings are not detected under the shelter of trees. This method can distinguish buildings from their adjacent open spaces and can still identify buildings in the case of tree interference.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>In order to quantitatively analyze the detection effect of our method and other excellent networks, we select the building detection network or semantic segmentation network in recent 3&#xa0;years as the comparison network of this paper, namely, BOMSC-Net (<xref ref-type="bibr" rid="B46">Zhou, et al., 2022</xref>), BMFR-Net (<xref ref-type="bibr" rid="B27">Ran, et al., 2021</xref>), STT (<xref ref-type="bibr" rid="B8">Chen, et al., 2021a</xref>), SRI-Net (<xref ref-type="bibr" rid="B23">Liu, et al., 2019</xref>), DR-Net (<xref ref-type="bibr" rid="B9">Chen, et al., 2021b</xref>), RSR-Net (<xref ref-type="bibr" rid="B17">Huang, et al., 2022</xref>)and B-FGC-Net (<xref ref-type="bibr" rid="B33">Wang, et al., 2022</xref>).</p>
<sec id="s4-1">
<title>4.1 Evaluation and comparisons</title>
<p>BOMSC-Net proposes a Multi-Scale Context Awareness Module (MSCAM) and a Direction Feature Optimization Module (DOM) by combining boundary optimization and multi-scale context awareness for problems such as tree and shadow occlusion and complex building roof materials. BMFR-Net combines Continuous Atrous Convolution Pyramid (CACP) module and Multi-scale Output Fusion Constraint (MOFC) for building detection. The two-way path conversion module is proposed in the Self-Service Terminal (SST) network, which can learn the long-term dependence features in space and channel dimensions and obtain more accurate building features. Spatial Residual Inception Network (SRI-Net) introduces deeply separable convolution and convolution decomposition, significantly reducing the number of model parameters while retaining global morphological features and local details. It makes the model lighter and more accurate in extracting building features. Dual-Rotation Network (DR-Net) combines densely connected convolutional neural network (DCNN) and residual network (ResNet) structures to extract buildings. RSR-Net improves the model&#x2019;s performance by introducing the SE attention module to reduce the noise effect of shallow features in feature fusion. B-FGC-Net optimizes network training by introducing residual learning and spatial attention units, highlighting the spatial information representation of features.</p>
<p>As mentioned above, the networks use feature pyramids and attention mechanism fusion methods, which are close to our method, so they are selected as the comparative experimental method in this paper. The experimental evaluation indicators are Precision, F-score, Recall, and IoU. Four indicators judge the ability of the network model from different aspects. Precision and Recall can measure the ability of the network to distinguish between false and correct targets. IoU and F-score can evaluate the overall accuracy of the network model.</p>
<p>The above four indicators data are from the original paper for fairness. &#x201c;-&#x201d; indicates that the data of this indicator is not provided in the article. The bolded indicator data indicates that the indicator is the highest, and the underline indicates that the indicator is the second. Quantitative evaluations are shown in the table below.</p>
<p>As shown in <xref ref-type="table" rid="T1">Table 1</xref>, our method&#x2019;s Precision, IoU, Recall, and F-score are 95.17%, 90.18%, 94.51%, and 94.83%, respectively. The accuracy of our method is lower than that of SRI-Net, indicating that the ability to detect the correct target is slightly lower than that of SRI-Net. However, the remaining three indicators in this paper are higher than that, indicating that our method is better than SRI-Net in distinguishing between correct and incorrect targets. Except for SRI-Net, the four indicators in this paper are higher than the above networks, which is enough to prove the advantages of our method in building detection. Therefore, our proposed method can more accurately detect buildings in remote sensing images.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of experimental results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">Precision (%)</th>
<th align="center">F-score (%)</th>
<th align="center">Recall (%)</th>
<th align="center">IoU (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">BOMSC-Net</td>
<td align="center">95.14</td>
<td align="center">94.80</td>
<td align="center">94.50</td>
<td align="center">90.15</td>
</tr>
<tr>
<td align="center">BMFR-Net</td>
<td align="center">94.31</td>
<td align="center">93.95</td>
<td align="center">94.42</td>
<td align="center">89.32</td>
</tr>
<tr>
<td align="center">RSR-Net</td>
<td align="center">94.92</td>
<td align="center">-</td>
<td align="center">92.63</td>
<td align="center">88.32</td>
</tr>
<tr>
<td align="center">B-FGC-Net</td>
<td align="center">95.03</td>
<td align="center">94.76</td>
<td align="center">94.49</td>
<td align="center">90.04</td>
</tr>
<tr>
<td align="center">STT</td>
<td align="center">-</td>
<td align="center">94.13</td>
<td align="center">-</td>
<td align="center">89.01</td>
</tr>
<tr>
<td align="center">SRI-Net</td>
<td align="center">
<bold>95.21</bold>
</td>
<td align="center">94.23</td>
<td align="center">93.28</td>
<td align="center">89.09</td>
</tr>
<tr>
<td align="center">DR-Net</td>
<td align="center">-</td>
<td align="center">93.80</td>
<td align="center">-</td>
<td align="center">88.30</td>
</tr>
<tr>
<td align="center">Ours</td>
<td align="center">95.17</td>
<td align="center">
<bold>94.83</bold>
</td>
<td align="center">
<bold>94.51</bold>
</td>
<td align="center">
<bold>90.18</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bolded text in table is meant to highlight the maximum values for each of the evaluation indicators.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-2">
<title>4.2 Ablation studies</title>
<p>In order to verify the effectiveness of the L-CAFSFM and MFFM in this paper, ablation experiments are designed in this paper. The experimental models are divided into BDRAR (<xref ref-type="bibr" rid="B16">Hou, et al., 2022</xref>), the network with only the L-CAFSFM, the network with only the MFF module, and the network in this paper. For fairness, all training and testing data configurations are the same. This paper evaluates the four network models from two aspects of qualitative analysis and quantitative calculation. Qualitative analysis and results are shown in <xref ref-type="fig" rid="F10">Figure 10</xref>.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Results of ablation experiments. The first column is the label map, the second column is the resulting map of the BDRAR model, the third column is the resulting map of only MFFM, the fourth column is the resulting map of only the L-CAFSFM and the fifth column is the experimental results of the method. The red part in the Figure is the missed detection area, the blue part is the false detection area, and the other parts are the same as the label map, which is the positive detection area.</p>
</caption>
<graphic xlink:href="feart-11-1268628-g010.tif"/>
</fig>
<p>In the above Figure, it can be seen from the second column that the missed detection rate of the BDRAR model is high. In the third column, it can be seen that after adding only the MFF module, the missed detection area decreases. It can be seen from the fourth column that after only adding the L-CAFSFM, the missed detection area is greatly reduced, but at the same time, the false detection area also increases. In the fifth column, that after adding L-CAFSFM and MFFM, compared with using the BDRAR model, the false detection area is slightly increased, but the missed detection area is greatly reduced.</p>
<p>The quantitative assessment of the ablation experiments is shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Results of ablation experiments.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">Precision (%)</th>
<th align="center">F-score (%)</th>
<th align="center">Recall (%)</th>
<th align="center">IoU (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">BDRAR</td>
<td align="center">93.91</td>
<td align="center">93.67</td>
<td align="center">93.42</td>
<td align="center">88.09</td>
</tr>
<tr>
<td align="center">Only L-CAFSFM</td>
<td align="center">94.94</td>
<td align="center">94.32</td>
<td align="center">93.70</td>
<td align="center">89.25</td>
</tr>
<tr>
<td align="center">Only MFFM</td>
<td align="center">94.64</td>
<td align="center">94.13</td>
<td align="center">94.13</td>
<td align="center">89.37</td>
</tr>
<tr>
<td align="center">
<bold>L-CAFSFM&#x2b; MFFM</bold>
</td>
<td align="center">
<bold>95.17</bold>
</td>
<td align="center">
<bold>94.83</bold>
</td>
<td align="center">
<bold>94.51</bold>
</td>
<td align="center">
<bold>90.18</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bolded text in table is meant to highlight the maximum values for each of the evaluation indicators.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>It can be seen from <xref ref-type="table" rid="T2">Table 2</xref> that after adding only the L-CAFSFM, the prediction accuracy and F-score are greatly improved, and the recall rate and IoU are only slightly improved. After only adding the MFF, all four indicators are improved, but the recall rate and IOU are greatly improved. After adding L-CAFSFM and MFFM, the four indicators have been greatly improved, consistent with the conclusions in the above experimental results. Therefore, it can be proved that L-CAFSFM and MFFM proposed in this paper have sound effects.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this article, we propose a building extraction method of combining L-CAFSFM, MFFM with a bidirectional feature pyramid network. L-CAFSFM calculates and fuses the feature maps of two adjacent levels to extract finer building details. The bidirectional feature pyramid network iteratively calculates L-CAFSFM and gradually learns multi-level feature information from two different directions to obtain fine features at different levels. MFFM integrates outputs from two directions to complement building feature information. The results are optimized using a dense conditional random field. Through the above improvements, the ability of the method to obtain rich and specific spatial features is further enhanced. Our method achieves state-of-the-art performance compared to other advanced models on the WHU dataset. The combination of L-CAFSFM and MFFM still has excellent potential to be applied in the field of computer vision, and we will continue to learn how to better apply L-CAFSFM and MFFM to building detection in the future.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="http://study.rsgis.whu.edu.cn/pages/download/">http://study.rsgis.whu.edu.cn/pages/download/</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>WF: Conceptualization, Methodology, Validation, Writing&#x2013;review and editing. FY: Conceptualization, Formal analysis, Methodology, Writing&#x2013;review and editing. XC: Conceptualization, Formal analysis, Methodology, Writing&#x2013;review and editing.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdollahi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pradhan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gite</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alamri</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Building footprint extraction from high resolution aerial images using generative adversarial network (gan) architecture</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>209517</fpage>&#x2013;<lpage>209527</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.3038225</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bauchet</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Mapurisa</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gobbin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tripodi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tarabalka</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Rooftops or footprints? Reliable building footprint extraction from high-resolution satellite images</article-title>. <source>IEEE Int. Geoscience Remote Sens. Symposium</source>, <fpage>274</fpage>&#x2013;<lpage>277</lpage>. <pub-id pub-id-type="doi">10.1109/IGARSS47720.2021.9554755</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Borba</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Diniz</surname>
<given-names>F. D. C.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>N. C. D.</given-names>
</name>
<name>
<surname>Bias</surname>
<given-names>E. d. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Building footprint extraction using deep learning semantic segmentation techniques: experiments and results</article-title>. <source>IEEE Int. Geoscience Remote Sens. Symposium</source>, <fpage>4708</fpage>&#x2013;<lpage>4711</lpage>. <pub-id pub-id-type="doi">10.1109/IGARSS47720.2021.9553855</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>MHA-net: multipath hybrid attention network for building footprint extraction from high-resolution remote sensing imagery</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Observations Remote Sens.</source> <volume>14</volume>, <fpage>5807</fpage>&#x2013;<lpage>5817</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2021.3084805</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Multi-scale building instance extraction framework in high resolution remote sensing imagery based on feature pyramid object-aware convolution neural network</article-title>,&#x201d; in <conf-name>Proceedings of the 2021 IEEE International Geoscience and Remote Sensing Symposium IGARSS</conf-name>, <conf-loc>Brussels, Belgium</conf-loc>, <conf-date>July 2021</conf-date>, <fpage>2779</fpage>&#x2013;<lpage>2782</lpage>. <pub-id pub-id-type="doi">10.1109/IGARSS47720.2021.9554016</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Building extraction from remote sensing images with conditional generative adversarial networks</article-title>,&#x201d; in <conf-name>Proceedings of the 2022 7th International Conference on Signal and Image Processing (ICSIP)</conf-name>, <conf-loc>Suzhou, China</conf-loc>, <conf-date>July 2022</conf-date>, <fpage>655</fpage>&#x2013;<lpage>658</lpage>. <pub-id pub-id-type="doi">10.1109/ICSIP55141.2022.9886096</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>A context feature enhancement network for building extraction from high-resolution remote sensing imagery</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>2276</fpage>. <pub-id pub-id-type="doi">10.3390/rs14092276</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>Building extraction from remote sensing images with sparse token transformers</article-title>. <source>Remote Sens.</source> <volume>13</volume> (<issue>21</issue>), <fpage>4441</fpage>. <pub-id pub-id-type="doi">10.3390/rs13214441</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2021b</year>). <article-title>DR-Net: an improved network for building extraction from high resolution remote sensing image</article-title>. <source>Remote Sens.</source> <volume>13</volume> (<issue>2</issue>), <fpage>294</fpage>. <pub-id pub-id-type="doi">10.3390/rs13020294</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xuan</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>CGSANet: a contour-guided and local structure-aware encoder&#x2013;decoder network for accurate building extraction from very high-resolution remote sensing imagery</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Observations Remote Sens.</source> <volume>15</volume>, <fpage>1526</fpage>&#x2013;<lpage>1542</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2021.3139017</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Deformable convolutional networks</article-title>. <comment>Avaialble at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1703.06211">https://arxiv.org/abs/1703.06211</ext-link>.</comment>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Das</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chand</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>AttentionBuildNet for building extraction from aerial imagery</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Computing, Communication, and Intelligent Systems</conf-name>, <conf-loc>Greater Noida, India</conf-loc>, <conf-date>February 2021</conf-date>, <fpage>576</fpage>&#x2013;<lpage>580</lpage>. <pub-id pub-id-type="doi">10.1109/ICCCIS51004.2021.9397178</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Improved pseudomasks generation for weakly supervised building extraction from high-resolution remote sensing imagery</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Observations Remote Sens.</source> <volume>15</volume>, <fpage>1629</fpage>&#x2013;<lpage>1642</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2022.3144176</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Building extraction based on U-Net with an attention block and multiple losses</article-title>. <source>Remote Sens.</source> <volume>12</volume> (<issue>9</issue>), <fpage>1400</fpage>. <pub-id pub-id-type="doi">10.3390/rs12091400</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Coordinate attention for efficient mobile network design</article-title>,&#x201d; in <conf-name>Proceedings of the 2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Nashville, TN, USA</conf-loc>, <conf-date>June 2021</conf-date>, <fpage>13708</fpage>&#x2013;<lpage>13717</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01350</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Multi-scale residual network for building extraction from satellite remote sensing images</article-title>,&#x201d; in <conf-name>Proceedings of the IGARSS 2022 - 2022 IEEE International Geoscience and Remote Sensing Symposium</conf-name>, <conf-loc>Kuala Lumpur, Malaysia</conf-loc>, <conf-date>July 2022</conf-date>, <fpage>1348</fpage>&#x2013;<lpage>1351</lpage>. <pub-id pub-id-type="doi">10.1109/IGARSS46834.2022.9883509</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A lightweight network for building extraction from remote sensing images</article-title>. <source>IEEE Trans. Geoscience Remote Sens.</source> <volume>60</volume> (<issue>5614812</issue>), <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2021.3131331</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Jie</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Squeeze-and-Excitation networks</article-title>,&#x201d; in <conf-name>Proceedings of the 2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>, <conf-loc>Salt Lake City, UT, USA</conf-loc>, <conf-date>June 2018</conf-date>.</citation>
</ref>
<ref id="B19">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gallagher</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Deeply-supervised nets</article-title>. <comment>Avaialble at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1409.5185">https://arxiv.org/abs/1409.5185</ext-link>.</comment>
</citation>
</ref>
<ref id="B20">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020a</year>). &#x201c;<article-title>Automatic extraction of built-up areas for cities in China from GF-3 images based on improved residual U-Net network</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE International Geoscience and Remote Sensing Symposium</conf-name>, <conf-loc>Waikoloa, HI, USA</conf-loc>, <conf-date>October 2020</conf-date>, <fpage>4399</fpage>&#x2013;<lpage>4402</lpage>. <pub-id pub-id-type="doi">10.1109/IGARSS39084.2020.9324329</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X. X.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Building footprint generation by integrating convolution neural network with feature pairwise conditional random field (FPCRF</article-title>. <source>IEEE Trans. Geoscience Remote Sens.</source> <volume>58</volume> (<issue>11</issue>), <fpage>7502</fpage>&#x2013;<lpage>7519</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2020.2973720</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Object-based morphological building index for building extraction from high resolution remote sensing imagery</article-title>. <source>Acta Geod. Cartogr. Sinica</source> <volume>46</volume> (<issue>6</issue>), <fpage>724</fpage>&#x2013;<lpage>733</lpage>. <pub-id pub-id-type="doi">10.11947/j.AGCS.2017.20170068</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Building footprint extraction from high-resolution images via spatial residual inception convolutional neural network</article-title>. <source>Remote Sens.</source> <volume>11</volume> (<issue>7</issue>), <fpage>830</fpage>. <pub-id pub-id-type="doi">10.3390/rs11070830</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Swin transformer: hierarchical vision transformer using shifted windows</article-title>,&#x201d; in <conf-name>Proceedings of the 2021 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>, <conf-loc>Montreal, QC, Canada</conf-loc>, <conf-date>October 2021</conf-date>, <fpage>9992</fpage>&#x2013;<lpage>10002</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00986</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An automatic morphological attribute building extraction approach for satellite high spatial resolution imagery</article-title>. <source>Remote Sens.</source> <volume>11</volume> (<issue>3</issue>), <fpage>337</fpage>. <pub-id pub-id-type="doi">10.3390/rs11030337</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Woo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J. Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>BAM: bottleneck attention module</article-title>. <comment>Avaialble at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1807.06514">https://arxiv.org/abs/1807.06514</ext-link>.</comment>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ran</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Building multi-feature fusion refined network for building extraction from high-resolution remote sensing images</article-title>. <source>Remote Sens.</source> <volume>13</volume> (<issue>14</issue>), <fpage>2794</fpage>. <pub-id pub-id-type="doi">10.3390/rs13142794</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Saini</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dixit</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Prajapati</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kushwaha</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Specific structure building extraction from high resolution satellite image</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Advances in Computing, Communication Control and Networking</conf-name>, <conf-loc>Greater Noida, India</conf-loc>, <conf-date>December 2021</conf-date>, <fpage>486</fpage>&#x2013;<lpage>489</lpage>. <pub-id pub-id-type="doi">10.1109/ICAC3N53548.2021.9725471</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheikh</surname>
<given-names>M. A. A.</given-names>
</name>
<name>
<surname>Maity</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kole</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>IRU-net: an efficient end-to-end network for automatic building extraction from remote sensing images</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>37811</fpage>&#x2013;<lpage>37828</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3164401</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>CSA-UNet: channel-spatial attention-based encoder&#x2013;decoder network for rural blue-roofed building extraction from UAV imagery</article-title>. <source>IEEE Geoscience Remote Sens. Lett.</source> <volume>15</volume>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/LGRS.2022.3197319</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multiscale building extraction with refined attention pyramid networks</article-title>. <source>IEEE Geoscience Remote Sens. Lett.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/LGRS.2021.3075436</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Building extraction from high&#x2013;resolution remote sensing images by adaptive morphological attribute profile under object boundary constraint</article-title>. <source>Sensors</source> <volume>19</volume> (<issue>17</issue>), <fpage>3737</fpage>. <pub-id pub-id-type="doi">10.3390/s19173737</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhuang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>B-FGC-Net: a building extraction network from high resolution remote sensing imagery</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>269</fpage>. <pub-id pub-id-type="doi">10.3390/rs14020269</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MBNet: multi-branch network for extraction of rural homesteads based on aerial images</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>2443</fpage>. <pub-id pub-id-type="doi">10.3390/rs14102443</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Woo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kweon</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>CBAM:Convolutional block attention module</article-title>. <comment>Avaialble at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1807.06521">https://arxiv.org/abs/1807.06521</ext-link>.</comment>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hui</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A Swin transformer-based encoding booster integrated in U-shaped network for building extraction</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>2611</fpage>. <pub-id pub-id-type="doi">10.3390/rs14112611</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Aggregated residual transformations for deep neural networks</article-title>,&#x201d; in <conf-name>Proceedings of the 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Honolulu, HI, USA</conf-loc>, <conf-date>July 2017</conf-date>, <fpage>5987</fpage>&#x2013;<lpage>5995</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.634</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Refined extraction of building outlines from high-resolution remote sensing imagery based on a multifeature convolutional neural network and morphological filtering</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Observations Remote Sens.</source> <volume>13</volume>, <fpage>1842</fpage>&#x2013;<lpage>1855</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2020.2991391</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>PANet: pixelwise affinity network for weakly supervised building extraction from high-resolution remote sensing images</article-title>. <source>IEEE Geoscience Remote Sens. Lett.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/LGRS.2022.3205309</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MAEANet: multiscale attention and edge-aware siamese network for building change detection in high-resolution remote sensing images</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>4895</fpage>. <pub-id pub-id-type="doi">10.3390/rs14194895</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>You</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>EfficientUNet&#x2b;: a building extraction method for emergency shelters based on deep learning</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>2207</fpage>. <pub-id pub-id-type="doi">10.3390/rs14092207</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Capsule feature pyramid network for building footprint extraction from high-resolution aerial imagery</article-title>. <source>IEEE Geoscience Remote Sens. Lett.</source> <volume>18</volume> (<issue>5</issue>), <fpage>895</fpage>&#x2013;<lpage>899</lpage>. <pub-id pub-id-type="doi">10.1109/lgrs.2020.2986380</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Mohd Shafri</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multi-modal feature fusion network with adaptive center point detector for building instance extraction</article-title>. <source>Remote Sens.</source> <volume>14</volume> (<issue>19</issue>), <fpage>4920</fpage>. <pub-id pub-id-type="doi">10.3390/rs14194920</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Srbuildingseg-E2: an integrated model for end-to-end higher-resolution building extraction</article-title>,&#x201d; in <conf-name>Proceedings of the IGARSS 2022 - 2022 IEEE International Geoscience and Remote Sensing Symposium</conf-name>, <conf-loc>Kuala Lumpur, Malaysia</conf-loc>, <conf-date>July 2022</conf-date>, <fpage>1356</fpage>&#x2013;<lpage>1359</lpage>. <pub-id pub-id-type="doi">10.1109/IGARSS46834.2022.9883295</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Building change detection using object-oriented LBP feature map in very high spatial resolution imagery</article-title>,&#x201d; in <conf-name>Proceedings of the 2019 10th International Workshop on the Analysis of Multitemporal Remote Sensing Images (MultiTemp)</conf-name>, <conf-loc>Shanghai, China</conf-loc>, <conf-date>August 2019</conf-date>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>BOMSC-net: boundary optimization and multi-scale context awareness based building extraction from high-resolution remote sensing imagery</article-title>. <source>IEEE Trans. Geoscience Remote Sens.</source> <volume>60</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2022.3152575</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Bidirectional feature pyramid network with recurrent attention residual modules for shadow detection</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>MAP-net: multiple attending path neural network for building footprint extraction from remote sensed imagery</article-title>. <source>IEEE Trans. Geoscience Remote Sens.</source> <volume>59</volume> (<issue>7</issue>), <fpage>6169</fpage>&#x2013;<lpage>6181</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2020.3026051</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>