<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurorobot.</journal-id>
<journal-title>Frontiers in Neurorobotics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurorobot.</abbrev-journal-title>
<issn pub-type="epub">1662-5218</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbot.2023.1267231</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Remote sensing traffic scene retrieval based on learning control algorithm for robot multimodal sensing information fusion and human-machine interaction and collaboration</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Peng</surname> <given-names>Huiling</given-names></name>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2389118/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Shi</surname> <given-names>Nianfeng</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2391841/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Guoqiang</given-names></name>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff><institution>School of Computer and Information Engineering, Luoyang Institute of Science and Technology</institution>, <addr-line>Luoyang</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Liping Zhang, Chinese Academy of Sciences (CAS), China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Ziheng Chen, Walmart Labs, United States; Sahraoui Dhelim, University College Dublin, Ireland; Jose Balsa Barreiro, Massachusetts Institute of Technology, United States; Ali Ya&#x001E7;ci, Karamano&#x001E7;lu Mehmetbey University, T&#x000FC;rkiye</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Huiling Peng <email>phl905&#x00040;lit.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>17</volume>
<elocation-id>1267231</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2023 Peng, Shi and Wang.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Peng, Shi and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license></permissions>
<abstract>
<p>In light of advancing socio-economic development and urban infrastructure, urban traffic congestion and accidents have become pressing issues. High-resolution remote sensing images are crucial for supporting urban geographic information systems (GIS), road planning, and vehicle navigation. Additionally, the emergence of robotics presents new possibilities for traffic management and road safety. This study introduces an innovative approach that combines attention mechanisms and robotic multimodal information fusion for retrieving traffic scenes from remote sensing images. Attention mechanisms focus on specific road and traffic features, reducing computation and enhancing detail capture. Graph neural algorithms improve scene retrieval accuracy. To achieve efficient traffic scene retrieval, a robot equipped with advanced sensing technology autonomously navigates urban environments, capturing high-accuracy, wide-coverage images. This facilitates comprehensive traffic databases and real-time traffic information retrieval for precise traffic management. Extensive experiments on large-scale remote sensing datasets demonstrate the feasibility and effectiveness of this approach. The integration of attention mechanisms, graph neural algorithms, and robotic multimodal information fusion enhances traffic scene retrieval, promising improved information extraction accuracy for more effective traffic management, road safety, and intelligent transportation systems. In conclusion, this interdisciplinary approach, combining attention mechanisms, graph neural algorithms, and robotic technology, represents significant progress in traffic scene retrieval from remote sensing images, with potential applications in traffic management, road safety, and urban planning.</p></abstract>
<kwd-group>
<kwd>remote sensing image processing (RSIP)</kwd>
<kwd>multimodal sensing</kwd>
<kwd>robot multimodality</kwd>
<kwd>sensor integration and fusion</kwd>
<kwd>multimodal information fusion</kwd>
</kwd-group>
<counts>
<fig-count count="10"/>
<table-count count="2"/>
<equation-count count="14"/>
<ref-count count="49"/>
<page-count count="15"/>
<word-count count="9282"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>The global trend of urbanization stands as one of the prominent features of contemporary world development. Over time, an increasing number of people are flocking to urban areas in search of improved living conditions and broader opportunities. However, this trend exhibits marked differences across regions. In some areas, such as Europe and North America, urbanization has been a historical process spanning several centuries, resulting in relatively mature urban planning and infrastructure development (Chaudhuri et al., <xref ref-type="bibr" rid="B7">2012</xref>). But, simultaneously, in some regions, especially in Asia and Africa, where development started later, the urbanization process is characterized by rapid growth, giving rise to a host of new challenges. This swift urbanization often leads to organizational issues stemming from inadequate urban planning, thereby impacting residents&#x00027; quality of life and the sustainability of urban development. China, especially in a global context, serves as a noteworthy case study. As one of the most populous countries globally, China has undergone an unprecedented pace and scale of urbanization, profoundly influencing global urbanization trends. In this global context, the fusion of multi-modal information has become increasingly crucial for decision-making (Plummer et al., <xref ref-type="bibr" rid="B33">2017</xref>). The integration of rapidly advancing remote sensing satellite technology with deep learning techniques has paved the way for significant advancements in various fields, including urban planning, disaster warning, and autonomous driving. The utilization of remote sensing data, acquired through spatial and spectral information obtained from digital image processing and analysis, has become an invaluable source for generating high-resolution satellite images (Liang et al., <xref ref-type="bibr" rid="B25">2017</xref>). These images play a crucial role in tasks such as target extraction, map updates, and geographical information system (GIS) information extraction (Chaudhuri et al., <xref ref-type="bibr" rid="B7">2012</xref>). However, extracting accurate road information from these high-resolution images has proven to be a challenging task using traditional methods that rely on grayscale feature analysis of image elements such as edge tracking or least squares B-spline curves. These methods suffer from issues related to accuracy, practicality, and generalizability. Therefore, this study aims to explore how to fully leverage remote sensing satellite technology and deep learning methods to address new challenges brought about by urbanization, especially in rapidly urbanizing regions like China.</p>
<p>To overcome these limitations, the integration of computer vision and deep learning (Wang et al., <xref ref-type="bibr" rid="B44">2022</xref>; Wu et al., <xref ref-type="bibr" rid="B46">2022</xref>, <xref ref-type="bibr" rid="B45">2023</xref>; Zhang M. et al., <xref ref-type="bibr" rid="B47">2022</xref>; Zhang Y.-H. et al., <xref ref-type="bibr" rid="B48">2022</xref>; Chen et al., <xref ref-type="bibr" rid="B9">2023</xref>) has become essential. Understanding the semantics of images has become increasingly important, and visual relationship detection has emerged as a key technique to improve computer understanding of images at a deeper level, providing high-level semantic information. In the early stages of visual relational research, common relationships between object pairs, such as position and size comparisons, were explored to improve object detection performance (Gaggioli et al., <xref ref-type="bibr" rid="B14">2016</xref>). However, recent advancements have expanded the scope to include spatial object-object interactions, prepositional and comparative adjective relations, and human object interactions (HOIS) (Gaggioli et al., <xref ref-type="bibr" rid="B14">2016</xref>), thereby enhancing the capabilities of vision tasks. Large-scale visual relationship detection has been achieved by decomposing the prediction of relationships into two parts: detecting objects and predicting predicates (Ben-Younes et al., <xref ref-type="bibr" rid="B2">2019</xref>). Fusing various visual features, such as appearance, size, bounding boxes, and linguistic cues, has been employed to build base phrases, contributing to better phrase localization (Plummer et al., <xref ref-type="bibr" rid="B33">2017</xref>). Furthermore, reinforcement learning frameworks and end-to-end systems have been proposed to enhance relationship detection through better object detection (Liang et al., <xref ref-type="bibr" rid="B25">2017</xref>; Rabbi et al., <xref ref-type="bibr" rid="B34">2020</xref>).</p>
<p>A combination of rich linguistic and visual representations has also been implemented using end-to-end deep neural networks, with the incorporation of external linguistic knowledge during training, leading to improved prediction and generalization (Kimura et al., <xref ref-type="bibr" rid="B19">2007</xref>). Deep learning, proposed by Chander et al. (<xref ref-type="bibr" rid="B6">2009</xref>), has been a game-changer in many fields, surpassing traditional machine learning methods by automatically learning from large datasets and uncovering implicit properties in the data. Since its success in the ImageNET competition in 2012, deep learning has become widely adopted in computer vision, speech recognition, natural language processing, medical image processing, and remote sensing, producing state-of-the-art results. As a result, the integration of deep learning models in remote sensing has significantly improved various remote sensing-related tasks, driving advancements in the field as a whole. The combination of rapidly advancing remote sensing satellite technology with deep learning techniques has enabled more accurate and effective decision-making through multi-modal information fusion. This fusion has opened up new possibilities for a wide range of applications, from urban planning and disaster warning to unmanned driving and GIS updates, propelling the field of remote sensing and its integration with computer vision to new heights. However, compared to computer vision images, remote sensing images present more challenges due to their larger coverage area, wider variety of objects, and complex backgrounds. As a result, when extracting information from remote sensing images, it becomes essential to employ deep learning models with visual attention mechanisms to effectively identify relevant features amidst the complexity, leading to improved accuracy in information extraction. The visual attention mechanism is a concept inspired by the human brain&#x00027;s ability to filter out relevant information (Tang, <xref ref-type="bibr" rid="B38">2022</xref>; Zheng et al., <xref ref-type="bibr" rid="B49">2022</xref>) from vast visual input. By applying a global quick scan followed by a focus on specific regions, the visual attention mechanism helps humans efficiently process external visual information. Integrating this mechanism into deep learning models simulates the human visual system&#x00027;s way of handling external information, making it a significant aspect to study for enhancing automatic information extraction from ultra-high resolution remote sensing images.</p>
<p>A more comprehensive elaboration of the research problem is indeed essential. The current introduction briefly mentions the challenges associated with traditional methods. Expanding on why these existing methods are inadequate, elucidating their limitations, and outlining why the integration of computer vision and deep learning is imperative will establish a robust foundation for this study (Liu et al., <xref ref-type="bibr" rid="B27">2017</xref>). Traditional methods often rely on manual data collection and feature engineering, which can be labor-intensive, time-consuming, and susceptible to human error. These approaches may struggle to cope with the ever-increasing volumes of complex visual data generated in various fields. Moreover, traditional methods may not adapt well to dynamic environments or effectively handle subtle nuances and variations in data. In contrast, computer vision and deep learning techniques have demonstrated their ability to automatically extract meaningful features from raw data, recognize patterns, and adapt to changing scenarios. Their potential to revolutionize tasks such as image classification, object detection, and video analysis has made them indispensable in modern research and applications. By delving into the limitations of conventional methods and highlighting the advantages of integrating computer vision and deep learning, we establish a compelling rationale for the necessity of our study. In summary, this study aims to accurately extract road information from high-resolution remote sensing images by combining attention mechanism fusion and Graph Neural Networks. The proposed approach shows significant promise in enhancing information extraction (Gao et al., <xref ref-type="bibr" rid="B15">2016</xref>; Liu et al., <xref ref-type="bibr" rid="B27">2017</xref>) and ultimately contributing to more effective traffic management, safer roads, and intelligent transportation systems. The interdisciplinary nature of this research, linking remote sensing, robotics, and transportation, opens up new research possibilities and applications in traffic management, road safety, and urban planning (<xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the overall structure of the research). The first part mainly introduces the development status of road information extraction based on remote sensing images, and describes the specific applications of road extraction methods at home and abroad, and lists the research objectives and significance of this paper, and explains the overall structure of this paper; the second part mainly introduces the related work, analyzes some of the most commonly used single pose estimation algorithms nowadays; the third part introduces the related algorithms used in this paper and the specific The fourth part describes the experimental process, which is based on the sample identification of the self-built dataset and the application of the Graph Neural Networks model algorithm with improved loss function, and verifies the validity and applicability of the model on the validation set, and compares some of the current mainstream algorithms. At the end of the paper, we discuss some advantages and disadvantages of the model, summarize the whole paper, and give an outlook on future work.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Schematic diagram of the overall structure of this paper, pre-processing of high-resolution remote sensing images, followed by extraction of road information from the images and model validation by combining Graph Neural Networks and dual attention mechanisms.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0001.tif"/>
</fig></sec>
<sec id="s2">
<title>2. Related work</title>
<p>In the context of robotic multimodal information fusion decision-making (Luo et al., <xref ref-type="bibr" rid="B28">2015</xref>), we can leverage advanced sensing technologies on robots to address the challenges in road extraction from remote sensing images (Mohd et al., <xref ref-type="bibr" rid="B31">2022</xref>). Robots equipped with various sensors can efficiently acquire data and analyze images, enhancing the accuracy and speed of road network extraction compared to manual visual interpretation methods. To achieve automatic or semi-automatic extraction of road networks, we can integrate the dynamic programming algorithm into the robot&#x00027;s decision-making process (Kubelka et al., <xref ref-type="bibr" rid="B20">2015</xref>). The dynamic programming algorithm, originally designed for low-resolution remote sensing images, has been improved for high-resolution images (Martins et al., <xref ref-type="bibr" rid="B30">2015</xref>).</p>
<p>By incorporating this algorithm into the robot&#x00027;s capabilities, we can develop an efficient method for road feature extraction (Lin et al., <xref ref-type="bibr" rid="B26">2020</xref>). The robot can use the dynamic programming algorithm to derive a parametric model of roads (Shi et al., <xref ref-type="bibr" rid="B37">2023</xref>), treating it as a &#x0201C;cost&#x0201D; function and applying dynamic programming to determine the optimal path between seed points. This approach allows the robot to iteratively optimize the &#x0201C;cost&#x0201D; function, incorporating constraints specific to road features, such as edge characteristics, to enhance accuracy (Li et al., <xref ref-type="bibr" rid="B22">2020</xref>). Moreover, the attention mechanism fusion can be applied to the robot&#x00027;s image analysis process. By integrating the attention mechanism into the extraction process, the robot can focus on specific road and traffic features, effectively reducing computation time while capturing relevant details for more accurate extraction results. Additionally, the use of graph neural algorithms can further enhance the robot&#x00027;s ability to detect and recognize road and traffic elements in the remote sensing images. These algorithms can process complex spatial relationships, improving the accuracy of scene retrieval.</p>
<p>With the robotic multimodal information fusion technique, the robot can autonomously navigate through urban environments and cover a wide area, capturing remote sensing images with high precision (Duan et al., <xref ref-type="bibr" rid="B13">2022</xref>). This data acquisition capability will contribute to creating comprehensive traffic databases and supporting real-time traffic information retrieval for more accurate traffic management and planning. In conclusion, by combining the dynamic programming algorithm with attention mechanism fusion and graph neural algorithms, integrated into robotic multimodal information (Tang et al., <xref ref-type="bibr" rid="B39">2021</xref>; He et al., <xref ref-type="bibr" rid="B17">2023</xref>) fusion decision-making, we can significantly improve the efficiency and accuracy of road extraction from remotely sensed images. This interdisciplinary approach bridges remote sensing, robotics, and transportation, opening up new research opportunities and applications in traffic management, road safety, and urban planning.</p>
<p>In the context of our proposed approach for traffic scene retrieval from remotely sensed images, we can leverage the advantages of the Snake model and visual attention mechanism to enhance the effectiveness of feature extraction and improve the accuracy of information retrieval.</p>
<p>Firstly, the Snake model, with its feature extraction capabilities and energy minimization process, can be employed to accurately extract road contours and traffic elements from high-resolution remote sensing images. However, to overcome its sensitivity to initial position and convergence issues, we can integrate the Snake model with the attention mechanism fusion. By simulating the visual attention mechanism, we can quickly identify areas of interest in the input image, helping to place the Snake model near the relevant image features of roads and traffic elements, thus improving its efficiency and precision in extracting the target regions (Valgaerts et al., <xref ref-type="bibr" rid="B42">2012</xref>). Moreover, the visual attention mechanism can play a crucial role in the process of compressing remote sensing image information. By analyzing human eye sensitivity to information, we can use different compression strategies based on the visual attention mechanism to compress remote sensing images effectively. This approach not only reduces the computational complexity but also improves the quality of the compressed images (Ghaffarian et al., <xref ref-type="bibr" rid="B16">2021</xref>).</p>
<p>Additionally, for target detection in remote sensing images, we can utilize the visual attention computational model to calculate saliency maps of the images. By identifying regions of interest using these saliency maps, we can achieve accurate target recognition and classification, which is beneficial for traffic scene retrieval and understanding (Dong et al., <xref ref-type="bibr" rid="B12">2021</xref>). Furthermore, semantic segmentation of remote sensing images can be improved by leveraging the visual attention mechanism. The attentional approach can assist in determining the exact boundaries of the targets to be extracted, leading to higher accuracy in information extraction from the remote sensing images (Buttar and Sachan, <xref ref-type="bibr" rid="B4">2022</xref>). Overall, the combination of the Snake model and visual attention mechanism in the context of our proposed approach can significantly enhance traffic scene retrieval from remotely sensed images. By effectively capturing relevant details, reducing computational burden, and improving the accuracy of feature extraction, this interdisciplinary approach holds promise for revolutionizing traffic management, road safety, and urban planning through intelligent transportation systems. The integration of robotic multimodal information fusion decision-making will further extend the capabilities of this approach by enabling advanced sensing technologies and autonomous navigation for data acquisition and analysis in urban environments (Valgaerts et al., <xref ref-type="bibr" rid="B42">2012</xref>; Dong et al., <xref ref-type="bibr" rid="B12">2021</xref>; Ghaffarian et al., <xref ref-type="bibr" rid="B16">2021</xref>; Buttar and Sachan, <xref ref-type="bibr" rid="B4">2022</xref>).</p>
<p>In our research, we acknowledge the effectiveness of deep learning in high-resolution remote sensing image information extraction. Deep convolutional neural network models have shown promising results in extracting valuable information from very high-resolution (VHR) images. The use of Fully Convolutional Networks (FCNs) for semantic segmentation has become the mainstream approach in this domain. Various studies (Maggiori et al., <xref ref-type="bibr" rid="B29">2017</xref>; Audebert et al., <xref ref-type="bibr" rid="B1">2018</xref>; Bittner et al., <xref ref-type="bibr" rid="B3">2018</xref>; Kampffmeyer et al., <xref ref-type="bibr" rid="B18">2018</xref>; Li et al., <xref ref-type="bibr" rid="B24">2018</xref>, <xref ref-type="bibr" rid="B21">2023</xref>; Shahzad et al., <xref ref-type="bibr" rid="B36">2018</xref>; Papadomanolaki et al., <xref ref-type="bibr" rid="B32">2019</xref>; Razi et al., <xref ref-type="bibr" rid="B35">2022</xref>) have demonstrated the benefits of employing deep learning methods to improve accuracy and feature representation in remote sensing image analysis.</p>
<p>Despite the progress made, the existing pixel-based methods can only reflect spectral information at an individual pixel level and lack a comprehensive understanding of the overall remote sensing image, leading to difficulties in obtaining meaningful object information and sensitivity to noise. Object-oriented and visual attention-based methods have shown potential but are limited by manual feature extraction and model robustness issues. Here, we propose a novel approach that incorporates attention mechanism fusion and robotic multimodal information fusion decision-making in the framework of graph neural algorithms to address these challenges (Chaib et al., <xref ref-type="bibr" rid="B5">2022</xref>; Chen et al., <xref ref-type="bibr" rid="B10">2022</xref>; Tian et al., <xref ref-type="bibr" rid="B41">2023</xref>).</p>
<p>Our approach utilizes attention mechanisms to enhance the focus on specific road and traffic features in the remotely sensed images. By doing so, we effectively reduce parameter computation and improve the ability to capture relevant details. We further employ graph neural algorithms to enhance the accuracy of scene retrieval, enabling more precise detection and recognition of road and traffic elements. The integration of robotic multimodal information fusion brings a new dimension to the process. Robots equipped with advanced sensing technologies can autonomously navigate urban environments and capture high-accuracy, wide-coverage remotely sensed images. By leveraging this multimodal information, we create comprehensive traffic databases that facilitate real-time traffic information retrieval, contributing to more accurate traffic management and planning.</p>
<p>Through extensive experiments on large-scale remote sensing datasets, we demonstrate the feasibility and effectiveness of our proposed approach. The combination of attentional mechanism fusion, graph neural algorithms, and robotic multimodal information fusion enhances the retrieval of traffic scenes from remotely sensed images, resulting in improved accuracy and efficiency of information extraction (Wang et al., <xref ref-type="bibr" rid="B43">2020</xref>; Tian and Ramdas, <xref ref-type="bibr" rid="B40">2021</xref>). Ultimately, this leads to more effective traffic management, safer roads, and intelligent transportation systems (Chen et al., <xref ref-type="bibr" rid="B8">2019</xref>; Cui et al., <xref ref-type="bibr" rid="B11">2019</xref>; Li and Zhu, <xref ref-type="bibr" rid="B23">2021</xref>). In conclusion, our interdisciplinary approach, which combines attentional mechanism fusion and robotic multimodal information fusion decision-making within the context of graph neural algorithms, presents significant advancements in retrieving traffic scenes from remotely sensed images. This integration of remote sensing, robotics, and transportation research opens up new possibilities for traffic management, road safety, and urban planning applications.</p></sec>
<sec id="s3">
<title>3. Method</title>
<sec>
<title>3.1. Road feature extraction</title>
<p>The spatial features of roads in remote sensing images are mainly manifested as one or more narrow and long curves, while the differences between road information and other objects are mainly reflected in texture and spatial features. Based on such differences between different objects, the typical features of roads can be effectively extracted and input into Alex network for model training, and the typical features of roads in sample data can be derived. In this paper, three methods are mainly used to extract road features in remote sensing images, including rectangular matching degree, linear feature index and second-order rectangular features.</p>
<sec>
<title>3.1.1. Rectangular matching degree</title>
<p>Since the road is a two-way lane in the actual environment, the interval between the two curves is stable and constant under normal circumstances, they are parallel to each other with a short interval, which is expressed as two parallel lines in the remote sensing image. The parameter <italic>M</italic><sub><italic>R</italic></sub> in the rectangular matching degree in the road extraction is used as the feature parameter of the road object, this is for this feature of the road in the remote sensing image, <italic>M</italic><sub><italic>R</italic></sub> is expressed as:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where: <italic>X</italic><sub><italic>in</italic></sub> indicates the area of the fitted rectangle of the image object; the larger the parameter <italic>M</italic><sub><italic>R</italic></sub> the more likely it is to be a road, and <italic>vice versa</italic>, the less likely it is to be a road; <italic>X</italic><sub><italic>oj</italic></sub> indicates the area of the acquired segmented image object.</p></sec>
<sec>
<title>3.1.2. RecLinear characteristic index</title>
<p>It is difficult to express road features directly by spectral features in high resolution remote sensing images, traverse and calculate the minimum outer rectangle of all segmented image objects, and the rectangle satisfies:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The linear characteristic index is obtained by calculating the centerline of all connected areas:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>F</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where, <italic>N</italic><sub><italic>P</italic></sub> denotes the area of the connected area; <italic>I</italic><sub><italic>LF</italic></sub> denotes the linear feature index; L and W denote the length and width of the image object; the larger the linear feature index, the more likely it is to be a road, and vice versa, the less likely it is to be a road.</p></sec>
<sec>
<title>3.1.3. Second-order rectangular-like features</title>
<p>In high-resolution remote sensing images, for regular roads, using rectangular matching degree to express the feature parameters with linear feature indicators will work well, but like some complex roads similar to loops, it is more difficult to use rectangular matching degree to express the feature parameters with linear feature indicators. To address this problem, a new linear feature, i.e., the second-order rectangular feature, is invoked in this paper to describe complex shapes:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mrow><mml:msub><mml:mi>C</mml:mi><mml:mi>M</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:mrow><mml:mi>n</mml:mi></mml:mfrac></mml:mrow></mml:math></disp-formula>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:msqrt></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>As shown in Equations (4) and (5) above, where <italic>F</italic><sub><italic>i</italic></sub> denotes the Euclidean distance between the ith image element and the center of mass inside the object; CM denotes the second-order moment characteristic parameter of the object; n denotes the number of image objects; i denotes the ith segmented object; <italic>X</italic><sub><italic>i</italic></sub>, <italic>Y</italic><sub><italic>i</italic></sub> denote the arbitrary image element of the ith object; <italic>X</italic><sub><italic>m</italic></sub>, <italic>y</italic><sub><italic>m</italic></sub>, denote the center of mass position of the ith object in the row and column directions.</p></sec></sec>
<sec>
<title>3.2. Graph Neural Networks</title>
<p>Graph Neural Networks (GNNs) are a type of deep learning model used for processing graph-structured data, and have achieved significant breakthroughs in fields such as image processing, social network analysis, and chemical molecule design. GNNs learn to represent the entire graph by iteratively propagating and aggregating information on the nodes and edges of the graph, enabling efficient processing of graph-structured data. The algorithmic principle of GNNs is as follows.The core idea of GNNs is to update the node representations through iterative information propagation and aggregation in the graph, starting from an initial feature representation for each node <inline-formula><mml:math id="M6"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> at time step <italic>t</italic> &#x0003D; 0, in an undirected graph <italic>G</italic> &#x0003D; (<italic>V, E</italic>) where <italic>V</italic> represents the set of nodes and <italic>E</italic> represents the set of edges. The update rule for each layer can be formalized as the following equation:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M8"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> represents the feature representation of node <italic>v</italic><sub><italic>i</italic></sub> at time step <italic>t</italic>, <italic>N</italic>(<italic>i</italic>) represents the set of neighboring nodes of node <italic>v</italic><sub><italic>i</italic></sub>, <italic>w</italic><sup>(<italic>t</italic>)</sup> represents the parameter set at time step <italic>t</italic>, and <italic>f</italic>(&#x000B7;) represents the update function for node features. This update function usually consists of two parts: information aggregation and activation function. Information aggregation updates the feature representation of the current node by aggregating the features of its neighboring nodes, which can be achieved through simple weighted averaging or more complex attention mechanisms. The activation function introduces non-linearity to increase the expressiveness of the model. After multiple layers of information propagation and aggregation, the final node feature representation <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> can be used for various tasks, such as node classification, graph classification, etc. For node classification tasks, the node features can be input into a fully connected layer followed by a softmax function for classification. For graph classification tasks, the features of all nodes are aggregated and input into a fully connected layer for classification, The structure of the Graph Neural Network is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Graph Neural Network structure schematic. A Graph Neural Network consists of an input layer, an output layer and multiple hidden layers. %: <bold>(A)</bold> Description of what is contained in the first panel. <bold>(B)</bold> Description of what is contained in the second panel.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0002.tif"/>
</fig>
<p>During the training process of GNNs, supervised learning methods are commonly used, optimizing the model parameters by minimizing the loss function between the predicted results and the true labels. The specific form of the loss function can vary depending on the task type and dataset, such as cross-entropy loss for node classification tasks, average pooling loss for graph classification tasks, etc. The algorithmic pseudo-code for the graphical convolutional neural network is shown in <xref ref-type="table" rid="T3">Algorithm 1</xref>.</p>
<table-wrap position="float" id="T3">
<label>Algorithm 1</label>
<caption><p>Graph Convolutional Neural Network (GCN).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-i0001.tif"/>
</table-wrap>
<p>In summary, GNNs learn to represent the entire graph by propagating and aggregating information on the nodes and edges of the graph, enabling efficient processing of graph-structured data.</p></sec>
<sec>
<title>3.3. Attention mechanism</title>
<p>Convolutional neural networks mimic the mechanism of biological visual perception, solving the tedious engineering of traditional manual feature extraction and realizing automatic feature extraction from data. However, the excellent performance of convolutional neural network is based on a large number of training samples, and the performance of convolutional neural network decreases dramatically under small samples because the training data is small and the sample features cannot be extracted comprehensively and effectively, so the attention mechanism is introduced into the convolutional neural network.</p>
<p>Visual attention mechanism as an important feature of human visual system. Combining attention mechanism with deep learning models, attention mechanism can help deep learning models to better understand external information. In deep learning attention mechanism is an effective tool to extract the most useful information from the input signal. Attention mechanisms usually use higher-level semantic information to reweight lower-level information to suppress background and noise. The attention mechanism is implemented by using filtering functions (e.g., softmax and sigmoid) and sequential techniques. Attention mechanisms combined with deep learning models have been applied to target detection, natural language processing, image classification, semantic segmentation, etc. with some success. <xref ref-type="fig" rid="F3">Figure 3</xref> displays a more complex attention mechanism, represented by a two-dimensional matrix that depicts the attention weights between Query and Key pairs. Each element in the matrix represents the attention weight between a Query and Key pair. The darker colors indicate higher attention weights, while lighter colors indicate lower attention weights.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Heat map of attention mechanism.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0003.tif"/>
</fig>
<p>The attention mechanism mimics the human visual attention pattern, focusing on only the most relevant originating information to the current task at a time, making the information request more efficient. The combination of attention mechanism and convolutional neural network can make the network model pay more attention to important features, increase the weight of important features, and suppress unnecessary features to further improve the feature extraction ability of convolutional neural network for important information. In this paper, a dual attention mechanism combining channel attention mechanism and spatial attention mechanism is introduced to combine Graph Neural Networks with a custom sample data set for more accurate detection and recognition of roads and transportation tools.</p>
<sec>
<title>3.3.1. Channel attention mechanism</title>
<p>The channel attention mechanism is based on a basic understanding of convolutional neural networks: features of different parts of an object are encoded on different channels of the convolutional feature map. The basic idea of the channel attention mechanism is to continuously adjust the weights of each channel through learning, and generate a vector of length equal to the number of channels through the network, and each element in the vector corresponds to the weight of each channel of the feature map, which in essence tells the network the parts of the pedestrian to be attended to. An attention mechanism algorithm pseudo-code is shown in <xref ref-type="table" rid="T4">Algorithm 2</xref>.</p>
<table-wrap position="float" id="T4">
<label>Algorithm 2</label>
<caption><p>The attention mechanism algorithm pseudo-code.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-i0002.tif"/>
</table-wrap>
<p>The network structure is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, where the classification branches are pooled first; the pooled weight vectors are fed into the fully connected layers FC1 and FC2 for &#x0201C;compression&#x0201D; and &#x0201C;stretching&#x0201D; operations. Then the components of the vectors are restricted between 0 and 1 by the sigmoid function, and the two vectors are summed and fused to form the final weight vector. In this paper, global pooling and maximum pooling are used simultaneously to highlight the main features while preserving the average characteristics of each channel, allowing the network to pay more attention to the visible parts of the pedestrians. The channel attention module generates a channel attention map using inter-channel relationships between features, and this feature assigns greater weight to channels where salient targets exhibit high response, as shown in the schematic diagram of the channel as attention module structure in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Channel attention sub-network structure.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0004.tif"/>
</fig>
<p>First, the input feature F is subjected to both maximum pooling and average pooling operations, and the null of the aggregated feature mapping The interval information is then input to a shared network, and the spatial dimension of the input feature map is compressed to sum the elements in the feature map one by one and generate the channel attention weights. The calculation formula is shown in equation below.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M32"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>A</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:mi>h</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mi>s</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>L</mml:mi><mml:mn>1</mml:mn><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></sec>
<sec>
<title>3.3.2. Spatial attention mechanism</title>
<p>Another attention mechanism cited in this paper is the spatial attention mechanism, which is essentially a network structure that generates a mask of the same size as the original image features, where the value of each element in the mask represents the feature map weight corresponding to the pixel at that location, and the weights change as they are continuously learned and adjusted, which in essence tells the network which regions to focus on. As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, the sub-network structure of the spatial attention mechanism in this paper.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Spatial attention sub-network structure.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0005.tif"/>
</fig>
<p>The feature map is first convolved by four 3 &#x000D7; 3-sized convolutional checks and return branches with both 256 channels, and then compressed into a single mask with 3 &#x000D7; 3 convolution and 1 channels. the original feature map is multiplied by EXP (mask parameter) multiplied to retain the original background information, thus adjusting the weights of each position of the original feature map. To guide the learning of the spatial attention mechanism, this paper uses the supervised information of the pedestrian as the label of the spatial attention mechanism to generate a pixel-level target mask: the pixel values of the pedestrian&#x00027;s full-body bounding box and visible bounding box regions are set to 0.8 and 1, respectively, and the pixel values of the remaining background regions are set to 0. Such a labeling will guide the spatial attention mechanism to focus its attention on the road regions in the frame at the The spatial attention mechanism is guided to pay more attention to the road visible area while focusing on the road area in the picture.</p></sec></sec>
<sec>
<title>3.4. Loss function</title>
<p>The loss function is an important part of the deep learning process, and the main purpose of using the loss function in this paper is to evaluate the prediction accuracy of the model and adjust the weight turnover, the loss function can make the convergence speed of the neural network become faster and make the prediction accuracy of the traffic scene more accurate. The most obvious difference between the two types of problems is that the result of regression prediction is continuous (e.g., house price), while the result of classification prediction is discrete (e.g., handwriting recognition). For example, semantic segmentation can be thought of as classifying each pixel in an image, and target detection can be thought of as regression on the position and size of a Bounding Box in an image.</p>
<sec>
<title>3.4.1. Overall loss function of the algorithm</title>
<p>In this paper, the parameters of each component are tuned jointly by a multi-task loss function, which consists of 3 components.</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M33"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>A</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where: <inline-formula><mml:math id="M34"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is an improved classification loss function in the basic form of a weighted cross&#x02212;entropy loss function, whose main purpose is to improve the problem of extreme imbalance between positive and negative samples in the regression&#x02212;based traffic scene detection algorithm; <italic>M</italic><sup><italic>c</italic></sup> is the number of all predicted frames; <italic>M</italic><sup><italic>r</italic></sup> is the number of all predicted frames, considering only the part judged as foreground; <inline-formula><mml:math id="M35"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is the loss function of the spatial attention mechanism sub-network, which <italic>L</italic>(<italic>m, m</italic><sup>&#x0002A;</sup>) is the loss function of the spatial attention mechanism subnetwork, which is actually a cross&#x02212;entropy loss function based on each pixel of the mask; p and p denote the category probability of the nth predicted traffic frame and the corresponding actual category, respectively; <inline-formula><mml:math id="M36"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is a new regression loss function proposed in this paper, which can design the size of the weights independently according to different masking degrees, and its design ideas and details will be introduced below; m, <italic>m</italic><sup>&#x0002A;</sup> are the masks generated by the spatial attention mechanism and their corresponding mask labels, respectively; &#x003BB;<sub><italic>m</italic></sub>, <italic>m</italic><sub>1</sub> are the masks generated by the spatial attention mechanism and their corresponding mask labels; &#x003BB; and &#x003BB;<sub>2</sub> are the parameters used to balance the sub&#x02212;loss functions, and their values are both 1 in this paper.</p></sec>
<sec>
<title>3.4.2. Regression loss function for occlusion perception</title>
<p>In generic target detection, the classical regression loss function is the smooth L1 function, which takes the form:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M37"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>A</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>w</mml:mi><mml:mo>,</mml:mo><mml:mi>h</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mi>s</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>L</mml:mi><mml:mn>1</mml:mn><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The expected weight is between 0 and 1, and even for a perfectly correct prediction frame, its IOU with the visible area may be a smaller value. The overlap between the visible area of the road and the prediction frame is used to improve the occlusion problem, and the practice is to judge this prediction frame as a positive sample only when the IOU of the prediction frame with both the road boundary frame and the visible area boundary frame is greater than a fixed threshold.</p></sec></sec></sec>
<sec id="s4">
<title>4. Experiment</title>
<sec>
<title>4.1. Experimental data set</title>
<p>The AI-TOD aerial image dataset was selected for training. 700,621 object instances in 8 categories from 28,036 aerial images were included in AI-TOD. In addition, 1000 instances of data from the Inria aerial image dataset were used as the validation set for model validation, As shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. The dataset consists of 1171 3-channel images and corresponding 2-channel segmentation labels with a spatial resolution of 1m, and each image has a size of 1500&#x0002A;1500 pixels. The labels are binarized images with a road pixel value of 1 and a background pixel value of 0. The dataset is randomly divided into three groups,seventy percent for the training set, twenty percent for the test set, and ten percent for the validation set. To avoid overfitting, the image preprocessing was first performed using a sliding window cropping technique with a span of 256 pixels, As shown in <xref ref-type="fig" rid="F7">Figure 7</xref>; then standard data enhancement was performed, and all cropped images were cropped, scaled, randomly rotated, horizontally and vertically flipped and image color changed. Some of the example data in the dataset are shown below.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Remote sensing image map in the dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0006.tif"/>
</fig>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Target extraction area in remote sensing images.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0007.tif"/>
</fig></sec>
<sec>
<title>4.2. Experimental platform</title>
<p>This paper uses Pytorch 1.10.0 deep learning framework, the operating system environment is Windows 10, the GPU is NVIDIA GeForce GTX 1650, the system memory size is 24G, and the programming language is Python 3.8.0. The initial learning rate is 0.001, which decreases to 0.0001 at 200 rounds and decays to 0.00001 at 260 rounds. 260 rounds of network training are conducted, and the batchsize of each GPU is 32. The images are modified by image preprocessing. In this paper, we crop the images in the COCO dataset to 256 &#x000D7; 192 size, and then achieve the data enhancement effect by random flipping and random scaling.</p></sec>
<sec>
<title>4.3. Evaluation criteria</title>
<p>Remote sensing images are widely used in various fields such as urban planning, traffic management, and environmental monitoring. Road extraction from remote sensing images is an essential task that enables the identification and mapping of transportation networks. It helps in building precise and accurate geographic information systems and enhancing the performance of autonomous vehicles. However, the automatic road extraction methods face significant challenges due to the complex and diverse environments of remote sensing images. Therefore, it is necessary to evaluate the quality of the road extraction methods using appropriate metrics. In this context, recall rate and cross-merge ratio are universal evaluation metrics that are commonly used to assess the performance of remote sensing image road extraction methods. The recall rate is defined as the ratio of true positive predictions to the total number of actual positive samples, which represents the ability of the method to detect all the road pixels. It is given by the following formula:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M38"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where TP is the number of true positives, and FN is the number of false negatives.</p>
<p>On the other hand, the cross-merge ratio represents the accuracy of the method in identifying the actual road pixels. It is defined as the ratio of true positive predictions to the total number of predicted road pixels, including both true positives and false positives. It is given by the following formula:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M39"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>U</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where FP is the number of false positives.</p>
<p>Both recall rate and cross-merge ratio are important metrics for evaluating the performance of road extraction methods. However, they have their limitations. For example, the recall rate only considers the ability of the method to detect road pixels, but it does not measure the accuracy of the detection. Therefore, a method with a high recall rate may produce a large number of false positives. On the other hand, the cross-merge ratio only measures the accuracy of the predicted road pixels, but it does not consider the ability of the method to detect all the road pixels. Therefore, a method with a high cross-merge ratio may miss some road pixels.</p>
<p>To overcome these limitations, other evaluation metrics have been proposed, such as F1-score, precision, and accuracy. The F1-score is the harmonic mean of precision and recall, and it is given by the following formula:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M40"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>P</mml:mi><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>R</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where P is the precision, which is the ratio of true positive predictions to the total number of predicted positive samples, and it is given by the following formula:</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M41"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The accuracy is the ratio of the total number of correct predictions to the total number of predictions, and it is given by the following formula:</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M42"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></sec>
<sec>
<title>4.4. Analysis of experimental results</title>
<p>In order to verify the feasibility of the road segmentation model, the U-Net model based on the U-Net model and the improved model in this paper to implement the road-road extraction task for high-resolution remote sensing images, As shown in the <xref ref-type="fig" rid="F8">Figure 8</xref>. Among them, the parameters of the comparison network are set the same as the original method. The performance comparison of different models for road segmentation is given in <xref ref-type="table" rid="T1">Table 1</xref>. It can be seen in <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F10">10</xref> that: the recall rate of this model is improved by five percent compared with that of U-Net, which is more consistent with the real labels and has better recognition rate for roads in image. the cross-merge ratio of this model is improved by zero point eight percent compared with that of U-Net, It shows the superior performance of the model in road extraction.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Results of model application in real remote sensing road extraction.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0008.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Results of experimental models compared to other research models.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>Accuracy</bold></th>
<th valign="top" align="left"><bold>Recall</bold></th>
<th valign="top" align="left"><bold>Precision</bold></th>
<th valign="top" align="left"><bold>F1-score</bold></th>
<th valign="top" align="left"><bold>Val-IOU</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">U-Net (Shahzad et al., <xref ref-type="bibr" rid="B36">2018</xref>)</td>
<td valign="top" align="left">80.1</td>
<td valign="top" align="left">73.1</td>
<td valign="top" align="left">74.8</td>
<td valign="top" align="left">74.3</td>
<td valign="top" align="left">77.6</td>
</tr> <tr>
<td valign="top" align="left">VGG (Kampffmeyer et al., <xref ref-type="bibr" rid="B18">2018</xref>)</td>
<td valign="top" align="left">76.5</td>
<td valign="top" align="left">74.12</td>
<td valign="top" align="left">76.89</td>
<td valign="top" align="left">77.05</td>
<td valign="top" align="left">79.23</td>
</tr> <tr>
<td valign="top" align="left">CNN (Razi et al., <xref ref-type="bibr" rid="B35">2022</xref>)</td>
<td valign="top" align="left">83.1</td>
<td valign="top" align="left">73.25</td>
<td valign="top" align="left">76.85</td>
<td valign="top" align="left">75.66</td>
<td valign="top" align="left">76.21</td>
</tr> <tr>
<td valign="top" align="left">DCNN (Razi et al., <xref ref-type="bibr" rid="B35">2022</xref>)</td>
<td valign="top" align="left">84.2</td>
<td valign="top" align="left">75.21</td>
<td valign="top" align="left">77.82</td>
<td valign="top" align="left">76.23</td>
<td valign="top" align="left">77.85</td>
</tr> <tr>
<td valign="top" align="left">LinkNet (Papadomanolaki et al., <xref ref-type="bibr" rid="B32">2019</xref>)</td>
<td valign="top" align="left">84.51</td>
<td valign="top" align="left">73.57</td>
<td valign="top" align="left">75.61</td>
<td valign="top" align="left">76.23</td>
<td valign="top" align="left">79.44</td>
</tr> <tr>
<td valign="top" align="left">D-LinkNet (Papadomanolaki et al., <xref ref-type="bibr" rid="B32">2019</xref>)</td>
<td valign="top" align="left">85.2</td>
<td valign="top" align="left">74.02</td>
<td valign="top" align="left">76.13</td>
<td valign="top" align="left">76.87</td>
<td valign="top" align="left">80.32</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="left">87.13</td>
<td valign="top" align="left">75.68</td>
<td valign="top" align="left">77.34</td>
<td valign="top" align="left">77.17</td>
<td valign="top" align="left">82.23</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Comparison of precision and recall results values of the model in this paper with other network models.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0009.tif"/>
</fig>
<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>Comparison of F1-score and Val-IOU results values of the model in this paper with other network models.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-17-1267231-g0010.tif"/>
</fig>
<p>In order to compare the performance of different attention modules, this paper introduces several attention modules into GNN. The results are shown in <xref ref-type="table" rid="T2">Table 2</xref>. Except for the ECA attention module, all the other three attention modules gained in the U-Net network and achieved better performance than the baseline network. Better performance than the baseline network was achieved.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>U-Net comparison experiments with several different attention modules.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
<th valign="top" align="center"><bold>Val-IOU</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ECA-GNN (Shahzad et al., <xref ref-type="bibr" rid="B36">2018</xref>)</td>
<td valign="top" align="center">75.24</td>
<td valign="top" align="center">75.53</td>
<td valign="top" align="center">73.45</td>
<td valign="top" align="center">76.3</td>
</tr> <tr>
<td valign="top" align="left">GNN (Kampffmeyer et al., <xref ref-type="bibr" rid="B18">2018</xref>)</td>
<td valign="top" align="center">76.91</td>
<td valign="top" align="center">74.23</td>
<td valign="top" align="center">74.57</td>
<td valign="top" align="center">77.13</td>
</tr> <tr>
<td valign="top" align="left">CBAM-GNN (Razi et al., <xref ref-type="bibr" rid="B35">2022</xref>)</td>
<td valign="top" align="center">77.31</td>
<td valign="top" align="center">73.23</td>
<td valign="top" align="center">75.21</td>
<td valign="top" align="center">76.88</td>
</tr> <tr>
<td valign="top" align="left">SE-GNN (Razi et al., <xref ref-type="bibr" rid="B35">2022</xref>)</td>
<td valign="top" align="center">75.48</td>
<td valign="top" align="center">76.58</td>
<td valign="top" align="center">75.74</td>
<td valign="top" align="center">76.93</td>
</tr> <tr>
<td valign="top" align="left">GC-GNN-Net (Razi et al., <xref ref-type="bibr" rid="B35">2022</xref>)</td>
<td valign="top" align="center">77.82</td>
<td valign="top" align="center">74.75</td>
<td valign="top" align="center">74.89</td>
<td valign="top" align="center">76.65</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">78.49</td>
<td valign="top" align="center">75.85</td>
<td valign="top" align="center">76.43</td>
<td valign="top" align="center">77.98</td>
</tr>
</tbody>
</table>
</table-wrap></sec></sec>
<sec sec-type="discussion" id="s5">
<title>5. Discussion</title>
<p>In response to the challenges faced in deep learning-based high-resolution remote sensing image information extraction, there are several issues that need to be addressed, such as the lack of global contextual information, over-segmentation problems, and difficulties in fusion of multimodal data. Additionally, the computational complexity, long training periods, and limited information transfer in both vertical and horizontal directions pose further challenges. To overcome these issues and improve the accuracy of remote sensing image information extraction, our proposed approach that integrates global attention mechanisms and robot-assisted techniques holds great promise in addressing the challenges of deep learning-based high-resolution remote sensing image information extraction. By incorporating global attention, we enhance the perception of scenes, improve object classification accuracy, and overcome the lack of global contextual information. Furthermore, the spatial attention mechanism helps reduce over-segmentation problems and facilitates the fusion of multimodal data, resulting in more accurate and detailed feature extraction. Robots equipped with advanced sensors and cameras are invaluable in the remote sensing domain. Their capabilities in data acquisition, preprocessing, and feature extraction contribute significantly to improving the efficiency and accuracy of remote sensing image analysis. Moreover, robot assistance in data fusion and annotation reduces manual efforts and enhances overall extraction processes.</p>
<p>The interdisciplinary nature of our approach bridges the gap between remote sensing, robotics, and transportation, opening new possibilities for research and applications in traffic management, road safety, and urban planning. By leveraging the strengths of attention mechanisms and robot-assisted techniques, we can achieve enhanced information extraction, leading to more effective traffic management, safer roads, and intelligent transportation systems. This advancement in the field of high-resolution remote sensing image analysis and robotics will undoubtedly have important economic value and far-reaching research significance in various applications and industries.</p></sec>
<sec sec-type="conclusions" id="s6">
<title>6. Conclusion</title>
<p>In the context of robotics, the proposed approach holds significant potential to augment the capabilities of robotic systems in road and transportation management. By integrating remote sensing and deep learning technologies into robots, they can actively contribute to various tasks, including road monitoring, traffic flow analysis, and autonomous navigation. With the ability to extract road and transportation features from remote sensing images, robots can efficiently carry out tasks such as road inspection, traffic surveillance, and swift response to accidents. The enhanced retrieval of traffic scenes from remotely sensed images, facilitated by the attentional mechanism fusion and graph neural algorithms, provides robots with precise and up-to-date information for effective decision-making in real-time traffic management and planning.</p>
<p>This integration of robotics with remote sensing and deep learning not only improves the efficiency of traffic-related tasks but also enhances road safety and overall transportation systems. With robots capable of autonomously navigating urban environments and capturing high-resolution remotely sensed images, comprehensive traffic databases can be created, allowing for more accurate and informed traffic management strategies.</p>
<p>As robotics technology continues to advance, the potential for further advancements in the field of intelligent transportation becomes even more promising. The ongoing evolution of deep learning techniques and visual attention mechanisms will continue to shape the future of remote sensing information extraction, enabling robots to play an even more significant role in traffic management, road safety, and urban planning. However, it is essential to acknowledge that current remote sensing information extraction methods based on deep learning still heavily rely on large training datasets, which can be resource-intensive to produce. Future research should focus on addressing this challenge and finding innovative ways to integrate domain knowledge and manual expertise with deep learning models to reduce dependence on extensive sample sets. By combining human expertise with cutting-edge technologies, the potential for advancing intelligent transportation systems becomes even greater.</p></sec>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p></sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>HP: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Project administration, Resources, Supervision, Visualization, Writing&#x02014;original draft, Writing&#x02014;review and editing. NS: Conceptualization, Data curation, Formal analysis, Resources, Software, Supervision, Validation, Writing&#x02014;original draft. GW: Investigation, Supervision, Validation, Visualization, Writing&#x02014;original draft.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack><p>The authors thank the participants for their valuable time in data collecting.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Audebert</surname> <given-names>N.</given-names></name> <name><surname>Le Saux</surname> <given-names>B.</given-names></name> <name><surname>Lef&#x000E8;vre</surname> <given-names>S.</given-names></name></person-group> (<year>2018</year>). <article-title>Beyond RGB: very high resolution urban remote sensing with multimodal deep networks</article-title>. <source>ISPRS J. Photogram. Remote Sens.</source> <volume>140</volume>, <fpage>20</fpage>&#x02013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2017.11.011</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ben-younes</surname> <given-names>H.</given-names></name> <name><surname>Cadene</surname> <given-names>R.</given-names></name> <name><surname>Thome</surname> <given-names>N.</given-names></name> <name><surname>Cord</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;BLOCK: bilinear superdiagonal fusion for visual question answering and visual relationship detection,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence</source> (<publisher-loc>Palo Alto, CA</publisher-loc>: <publisher-name>Stanford University</publisher-name>), <fpage>8102</fpage>&#x02013;<lpage>8109</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bittner</surname> <given-names>K.</given-names></name> <name><surname>Adam</surname> <given-names>F.</given-names></name> <name><surname>Cui</surname> <given-names>S.</given-names></name> <name><surname>K&#x000F6;rner</surname> <given-names>M.</given-names></name> <name><surname>Reinartz</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>Building footprint extraction from VHR remote sensing images combined with normalized dsms using fused fully convolutional networks</article-title>. <source>IEEE J. Select. Top. Appl. Earth Observ. Remote Sens.</source> <volume>11</volume>, <fpage>2615</fpage>&#x02013;<lpage>2629</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2018.2849363</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buttar</surname> <given-names>P. K.</given-names></name> <name><surname>Sachan</surname> <given-names>M. K.</given-names></name></person-group> (<year>2022</year>). <article-title>Semantic segmentation of clouds in satellite images based on U-Net&#x0002B;&#x0002B; architecture and attention mechanism</article-title>. <source>Expert Syst. Appl.</source> <volume>209</volume>, <fpage>118380</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2022.118380</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chaib</surname> <given-names>S.</given-names></name> <name><surname>Mansouri</surname> <given-names>D. E. K.</given-names></name> <name><surname>Omara</surname> <given-names>I.</given-names></name> <name><surname>Hagag</surname> <given-names>A.</given-names></name> <name><surname>Dhelim</surname> <given-names>S.</given-names></name> <name><surname>Bensaber</surname> <given-names>D. A.</given-names></name></person-group> (<year>2022</year>). <article-title>On the co-selection of vision transformer features and images for very high-resolution image scene classification</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>5817</fpage>. <pub-id pub-id-type="doi">10.3390/rs14225817</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chander</surname> <given-names>G.</given-names></name> <name><surname>Markham</surname> <given-names>B. L.</given-names></name> <name><surname>Helder</surname> <given-names>D. L.</given-names></name></person-group> (<year>2009</year>). <article-title>Summary of current radiometric calibration coefficients for Landsat MSS, TM, ETM&#x0002B;, and EO-1 ALI sensors</article-title>. <source>Remote Sens. Environ.</source> <volume>113</volume>, <fpage>893</fpage>&#x02013;<lpage>903</lpage>. <pub-id pub-id-type="doi">10.1016/j.rse.2009.01.007</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chaudhuri</surname> <given-names>D.</given-names></name> <name><surname>Kushwaha</surname> <given-names>N. K.</given-names></name> <name><surname>Samal</surname> <given-names>A.</given-names></name></person-group> (<year>2012</year>). <article-title>Semi-automated road detection from high resolution satellite images by directional morphological enhancement and segmentation techniques</article-title>. <source>IEEE J. Select. Top. Appl. Earth Observ. Remote Sens.</source> <volume>5</volume>, <fpage>1538</fpage>&#x02013;<lpage>1544</lpage>. <pub-id pub-id-type="doi">10.1109/JSTARS.2012.2199085</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>C.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Teo</surname> <given-names>S. G.</given-names></name> <name><surname>Zou</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Gated residual recurrent graph neural networks for traffic prediction,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence</source> (<publisher-loc>Palo Alto, CA</publisher-loc>: <publisher-name>Stanford University</publisher-name>), <fpage>485</fpage>&#x02013;<lpage>492</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Wu</surname> <given-names>Y. C.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Sea ice extent prediction with machine learning methods and subregional analysis in the Arctic</article-title>. <source>Atmosphere</source> <volume>14</volume>, <fpage>1023</fpage>. <pub-id pub-id-type="doi">10.3390/atmos14061023</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Silvestri</surname> <given-names>F.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Zhu</surname> <given-names>H.</given-names></name> <name><surname>Ahn</surname> <given-names>H.</given-names></name> <name><surname>Tolomei</surname> <given-names>G.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Relax: reinforcement learning agent explainer for arbitrary predictive models,&#x0201D;</article-title> in <source>Proceedings of the 31st ACM International Conference on Information &#x00026; Knowledge Management</source> (<publisher-loc>Atlanta, GA</publisher-loc>), <fpage>252</fpage>&#x02013;<lpage>261</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>Z.</given-names></name> <name><surname>Henrickson</surname> <given-names>K.</given-names></name> <name><surname>Ke</surname> <given-names>R.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Traffic graph convolutional recurrent neural network: a deep learning framework for network-scale traffic learning and forecasting</article-title>. <source>IEEE Trans. Intell. Transport. Syst.</source> <volume>21</volume>, <fpage>4883</fpage>&#x02013;<lpage>4894</lpage>. <pub-id pub-id-type="doi">10.1109/TITS.2019.2950416</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Han</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>Ship object detection of remote sensing image based on visual attention</article-title>. <source>Remote Sens.</source> <volume>13</volume>, <fpage>3192</fpage>. <pub-id pub-id-type="doi">10.3390/rs13163192</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>S.</given-names></name> <name><surname>Shi</surname> <given-names>Q.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Multimodal sensors and ML-based data fusion for advanced robots</article-title>. <source>Adv. Intell. Syst.</source> <volume>4</volume>, <fpage>2200213</fpage>. <pub-id pub-id-type="doi">10.1002/aisy.202200213</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gaggioli</surname> <given-names>A.</given-names></name> <name><surname>Ferscha</surname> <given-names>A.</given-names></name> <name><surname>Riva</surname> <given-names>G.</given-names></name> <name><surname>Dunne</surname> <given-names>S.</given-names></name> <name><surname>Viaud-Delmon</surname> <given-names>I.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Human computer confluence,&#x0201D;</article-title> in <source>Human Computer Confluence</source> (<publisher-loc>De Gruyter Open Poland</publisher-loc>).</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>S.</given-names></name> <name><surname>Zang</surname> <given-names>P.</given-names></name> <name><surname>Dang</surname> <given-names>L.</given-names></name> <name><surname>Shi</surname> <given-names>F.</given-names></name> <name><surname>Xu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Polyaniline nanorods grown on hollow carbon fibers as high-performance supercapacitor electrodes</article-title>. <source>ChemElectroChem</source> <volume>3</volume>, <fpage>1142</fpage>&#x02013;<lpage>1149</lpage>. <pub-id pub-id-type="doi">10.1002/celc.201600153</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghaffarian</surname> <given-names>S.</given-names></name> <name><surname>Valente</surname> <given-names>J.</given-names></name> <name><surname>Van Der Voort</surname> <given-names>M.</given-names></name> <name><surname>Tekinerdogan</surname> <given-names>B.</given-names></name></person-group> (<year>2021</year>). <article-title>Effect of attention mechanism in deep learning-based remote sensing image processing: a systematic literature review</article-title>. <source>Remote Sens.</source> <volume>13</volume>, <fpage>2965</fpage>. <pub-id pub-id-type="doi">10.3390/rs13152965</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>F.</given-names></name> <name><surname>Chang</surname> <given-names>T.-L.</given-names></name> <name><surname>Tang</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Bacterial detection and differentiation of <italic>Staphylococcus aureus</italic> and <italic>Escherichia coli</italic> utilizing long-period fiber gratings functionalized with nanoporous coated structures</article-title>. <source>Coatings</source> <volume>13</volume>, <fpage>778</fpage>. <pub-id pub-id-type="doi">10.3390/coatings13040778</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kampffmeyer</surname> <given-names>M.</given-names></name> <name><surname>Dong</surname> <given-names>N.</given-names></name> <name><surname>Liang</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Xing</surname> <given-names>E. P.</given-names></name></person-group> (<year>2018</year>). <article-title>CONNNet: a long-range relation-aware pixel-connectivity network for salient segmentation</article-title>. <source>IEEE Trans. Image Process.</source> <volume>28</volume>, <fpage>2518</fpage>&#x02013;<lpage>2529</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2018.2886997</pub-id><pub-id pub-id-type="pmid">30571633</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kimura</surname> <given-names>R.</given-names></name> <name><surname>Bai</surname> <given-names>L.</given-names></name> <name><surname>Fan</surname> <given-names>J.</given-names></name> <name><surname>Takayama</surname> <given-names>N.</given-names></name> <name><surname>Hinokidani</surname> <given-names>O.</given-names></name></person-group> (<year>2007</year>). <article-title>Evapo-transpiration estimation over the river basin of the loess plateau of China based on remote sensing</article-title>. <source>J. Arid Environ.</source> <volume>68</volume>, <fpage>53</fpage>&#x02013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/j.jaridenv.2006.03.029</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kubelka</surname> <given-names>V.</given-names></name> <name><surname>Oswald</surname> <given-names>L.</given-names></name> <name><surname>Pomerleau</surname> <given-names>F.</given-names></name> <name><surname>Colas</surname> <given-names>F.</given-names></name> <name><surname>Svoboda</surname> <given-names>T.</given-names></name> <name><surname>Reinstein</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>Robust data fusion of multimodal sensory information for mobile robots</article-title>. <source>J. Field Robot.</source> <volume>32</volume>, <fpage>447</fpage>&#x02013;<lpage>473</lpage>. <pub-id pub-id-type="doi">10.1002/rob.21535</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>D.</given-names></name> <name><surname>Pang</surname> <given-names>B.</given-names></name> <name><surname>Lv</surname> <given-names>S.</given-names></name> <name><surname>Yin</surname> <given-names>Z.</given-names></name> <name><surname>Lian</surname> <given-names>X.</given-names></name> <name><surname>Sun</surname> <given-names>D.</given-names></name></person-group> (<year>2023</year>). <article-title>A double-layer feature fusion convolutional neural network for infrared small target detection</article-title>. <source>Int. J. Remote Sens.</source> <volume>44</volume>, <fpage>407</fpage>&#x02013;<lpage>427</lpage>. <pub-id pub-id-type="doi">10.1080/01431161.2022.2161852</pub-id><pub-id pub-id-type="pmid">35663726</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Peng</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Luo</surname> <given-names>Z.</given-names></name> <name><surname>Tang</surname> <given-names>C.</given-names></name></person-group> (<year>2020</year>). <article-title>Multimodal information fusion for automatic aesthetics evaluation of robotic dance poses</article-title>. <source>Int. J. Soc. Robot.</source> <volume>12</volume>, <fpage>5</fpage>&#x02013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1007/s12369-019-00535-w</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Zhu</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Spatial-temporal fusion graph neural networks for traffic flow forecasting,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence</source> (<publisher-loc>Palo Alto, CA</publisher-loc>: <publisher-name>Stanford University</publisher-name>), <fpage>4189</fpage>&#x02013;<lpage>4196</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Dong</surname> <given-names>R.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Yu</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Large-scale oil palm tree detection from high-resolution satellite images using two-stage convolutional neural networks</article-title>. <source>Remote Sens.</source> <volume>11</volume>, <fpage>11</fpage>. <pub-id pub-id-type="doi">10.3390/rs11010011</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>X.</given-names></name> <name><surname>Lee</surname> <given-names>L.</given-names></name> <name><surname>Xing</surname> <given-names>E. P.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Deep variation-structured reinforcement learning for visual relationship and attribute detection,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Long Beach</publisher-loc>), <fpage>848</fpage>&#x02013;<lpage>857</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>K.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Zhou</surname> <given-names>D.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name></person-group> (<year>2020</year>). <article-title>Multi-sensor fusion for body sensor network in medical human&#x02013;robot interaction scenario</article-title>. <source>Inform. Fusion</source> <volume>57</volume>, <fpage>15</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2019.11.001</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>He</surname> <given-names>S.</given-names></name> <name><surname>Robbins</surname> <given-names>J. P.</given-names></name> <name><surname>Podkolzin</surname> <given-names>S. G.</given-names></name> <name><surname>Tian</surname> <given-names>F.</given-names></name></person-group> (<year>2017</year>). <article-title>Observation and identification of an atomic oxygen structure on catalytic gold nanoparticles</article-title>. <source>Angew. Chem.</source> <volume>129</volume>, <fpage>13132</fpage>&#x02013;<lpage>13137</lpage>. <pub-id pub-id-type="doi">10.1002/ange.201706647</pub-id><pub-id pub-id-type="pmid">28776923</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>R. C.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>P.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Multimodal information fusion for human robot interaction,&#x0201D;</article-title> in <source>2015 IEEE 10th Jubilee International Symposium on Applied Computational Intelligence and Informatics</source> (<publisher-loc>Timisoara</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>535</fpage>&#x02013;<lpage>540</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maggiori</surname> <given-names>E.</given-names></name> <name><surname>Tarabalka</surname> <given-names>Y.</given-names></name> <name><surname>Charpiat</surname> <given-names>G.</given-names></name> <name><surname>Alliez</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>High-resolution aerial image labeling with convolutional neural networks</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>55</volume>, <fpage>7092</fpage>&#x02013;<lpage>7103</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2017.2740362</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Martins</surname> <given-names>&#x000C9;. F.</given-names></name> <name><surname>Dal Poz</surname> <given-names>A. P.</given-names></name> <name><surname>Gallis</surname> <given-names>R. A.</given-names></name></person-group> (<year>2015</year>). <article-title>Semiautomatic object-space road extraction combining a stereoscopic image pair and a tin-based DTM</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>12</volume>, <fpage>1790</fpage>&#x02013;<lpage>1794</lpage>. <pub-id pub-id-type="doi">10.1109/LGRS.2015.2426112</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mohd</surname> <given-names>T. K.</given-names></name> <name><surname>Nguyen</surname> <given-names>N.</given-names></name> <name><surname>Javaid</surname> <given-names>A. Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Multi-modal data fusion in enhancing human-machine interaction for robotic applications: a survey</article-title>. <source>arXiv preprint arXiv:2202.07732</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2202.07732</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Papadomanolaki</surname> <given-names>M.</given-names></name> <name><surname>Vakalopoulou</surname> <given-names>M.</given-names></name> <name><surname>Karantzalos</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>A novel object-based deep learning framework for semantic segmentation of very high-resolution remote sensing data: comparison with convolutional and fully convolutional networks</article-title>. <source>Remote Sens.</source> <volume>11</volume>, <fpage>684</fpage>. <pub-id pub-id-type="doi">10.3390/rs11060684</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Plummer</surname> <given-names>B. A.</given-names></name> <name><surname>Mallya</surname> <given-names>A.</given-names></name> <name><surname>Cervantes</surname> <given-names>C. M.</given-names></name> <name><surname>Hockenmaier</surname> <given-names>J.</given-names></name> <name><surname>Lazebnik</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Phrase localization and visual relationship detection with comprehensive image-language cues,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Honolulu</publisher-loc>), <fpage>1928</fpage>&#x02013;<lpage>1937</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rabbi</surname> <given-names>J.</given-names></name> <name><surname>Ray</surname> <given-names>N.</given-names></name> <name><surname>Schubert</surname> <given-names>M.</given-names></name> <name><surname>Chowdhury</surname> <given-names>S.</given-names></name> <name><surname>Chao</surname> <given-names>D.</given-names></name></person-group> (<year>2020</year>). <article-title>Small-object detection in remote sensing images with end-to-end edge-enhanced GAN and object detector network</article-title>. <source>Remote Sens.</source> <volume>12</volume>, <fpage>1432</fpage>. <pub-id pub-id-type="doi">10.3390/rs12091432</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Razi</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Russo</surname> <given-names>B.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Deep learning serves traffic safety analysis: a forward-looking review</article-title>. <source>IET Intell. Transport Syst.</source> <pub-id pub-id-type="doi">10.1049/itr2.12257</pub-id> [Epub ahead of print].</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shahzad</surname> <given-names>M.</given-names></name> <name><surname>Maurer</surname> <given-names>M.</given-names></name> <name><surname>Fraundorfer</surname> <given-names>F.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Zhu</surname> <given-names>X. X.</given-names></name></person-group> (<year>2018</year>). <article-title>Buildings detection in VHR SAR images using fully convolution neural networks</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>57</volume>, <fpage>1100</fpage>&#x02013;<lpage>1116</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2018.2864716</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>Q.</given-names></name> <name><surname>Sun</surname> <given-names>Z.</given-names></name> <name><surname>Le</surname> <given-names>X.</given-names></name> <name><surname>Xie</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>Soft robotic perception system with ultrasonic auto-positioning and multimodal sensory intelligence</article-title>. <source>ACS Nano</source> <volume>17</volume>, <fpage>4985</fpage>&#x02013;<lpage>4998</lpage>. <pub-id pub-id-type="doi">10.1021/acsnano.2c12592</pub-id><pub-id pub-id-type="pmid">36867760</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="thesis"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <source>Molecular fundamentals of upgrading biomass-derived feedstocks over platinum-molybdenum catalysts</source> (Ph.D. thesis). Stevens Institute of Technology, Hoboken, NJ, United States.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Du</surname> <given-names>H.</given-names></name> <name><surname>Podkolzin</surname> <given-names>S. G.</given-names></name></person-group> (<year>2021</year>). <article-title>Atomic, molecular and hybrid oxygen structures on silver</article-title>. <source>Langmuir</source> <volume>37</volume>, <fpage>11603</fpage>&#x02013;<lpage>11610</lpage>. <pub-id pub-id-type="doi">10.1021/acs.langmuir.1c01941</pub-id><pub-id pub-id-type="pmid">34565146</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>J.</given-names></name> <name><surname>Ramdas</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Online control of the familywise error rate</article-title>. <source>Stat. Methods Med. Res.</source> <volume>30</volume>, <fpage>976</fpage>&#x02013;<lpage>993</lpage>. <pub-id pub-id-type="doi">10.1177/0962280220983381</pub-id><pub-id pub-id-type="pmid">33413033</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>Y.</given-names></name> <name><surname>Carballo</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Takeda</surname> <given-names>K.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;RSG-search: semantic traffic scene retrieval using graph-based scene representation,&#x0201D;</article-title> in <source>2023 IEEE Intelligent Vehicles Symposium (IV)</source> (<publisher-loc>Aachen</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Valgaerts</surname> <given-names>L.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Bruhn</surname> <given-names>A.</given-names></name> <name><surname>Seidel</surname> <given-names>H.-P.</given-names></name> <name><surname>Theobalt</surname> <given-names>C.</given-names></name></person-group> (<year>2012</year>). <article-title>Lightweight binocular facial performance capture under uncontrolled lighting</article-title>. <source>ACM Trans. Graph.</source> <volume>31</volume>, <fpage>1</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1145/2366145.2366206</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Jin</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Traffic flow prediction via spatial temporal graph neural network,&#x0201D;</article-title> in <source>Proceedings of the Web Conference 2020</source> (<publisher-loc>Taipei</publisher-loc>), <fpage>1082</fpage>&#x02013;<lpage>1092</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Jian</surname> <given-names>Y.</given-names></name> <name><surname>Qureshi</surname> <given-names>S.</given-names></name> <name><surname>Jie</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>On the comparative use of social media data and survey data in prioritizing ecosystem services for cost-effective governance</article-title>. <source>Ecosyst. Serv.</source> <volume>56</volume>, <fpage>101446</fpage>. <pub-id pub-id-type="doi">10.1016/j.ecoser.2022.101446</pub-id></citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Pichler</surname> <given-names>D.</given-names></name> <name><surname>Marley</surname> <given-names>D.</given-names></name> <name><surname>Wilson</surname> <given-names>D.</given-names></name> <name><surname>Hovakimyan</surname> <given-names>N.</given-names></name> <name><surname>Hobbs</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Extended agriculture-vision: an extension of a large aerial image dataset for agricultural pattern analysis</article-title>. <source>arXiv preprint arXiv:2303.02460</source>.</citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Tao</surname> <given-names>R.</given-names></name> <name><surname>Zhao</surname> <given-names>P.</given-names></name> <name><surname>Martin</surname> <given-names>N. F.</given-names></name> <name><surname>Hovakimyan</surname> <given-names>N.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Optimizing nitrogen management with deep reinforcement learning and crop simulations,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>New Orleans</publisher-loc>), <fpage>1712</fpage>&#x02013;<lpage>1720</lpage>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Xie</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.-H.</given-names></name> <name><surname>Wen</surname> <given-names>C.</given-names></name> <name><surname>He</surname> <given-names>J.-B.</given-names></name></person-group> (<year>2022</year>). <article-title>Fine segmentation on faces with masks based on a multistep iterative segmentation algorithm</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>75742</fpage>&#x02013;<lpage>75753</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3192026</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.-H.</given-names></name> <name><surname>Wen</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Xie</surname> <given-names>K.</given-names></name> <name><surname>He</surname> <given-names>J.-B.</given-names></name></person-group> (<year>2022</year>). <article-title>Fast 3D visualization of massive geological data based on clustering index fusion</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>28821</fpage>&#x02013;<lpage>28831</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3157823</pub-id></citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>Y.</given-names></name> <name><surname>Qi</surname> <given-names>Y.</given-names></name> <name><surname>Tang</surname> <given-names>Z.</given-names></name> <name><surname>Tan</surname> <given-names>J.</given-names></name> <name><surname>Koel</surname> <given-names>B. E.</given-names></name> <name><surname>Podkolzin</surname> <given-names>S. G.</given-names></name></person-group> (<year>2022</year>). <article-title>Spectroscopic observation and structure-insensitivity of hydroxyls on gold</article-title>. <source>Chem. Commun.</source> <volume>58</volume>, <fpage>4036</fpage>&#x02013;<lpage>4039</lpage>. <pub-id pub-id-type="doi">10.1039/D2CC00283C</pub-id><pub-id pub-id-type="pmid">35258054</pub-id></citation></ref>
</ref-list>
</back>
</article>