<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurorobot.</journal-id>
<journal-title>Frontiers in Neurorobotics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurorobot.</abbrev-journal-title>
<issn pub-type="epub">1662-5218</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbot.2024.1489021</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Cascade contour-enhanced panoptic segmentation for robotic vision perception</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Yue</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2815576/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Runze</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2831823/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhu</surname> <given-names>Dongchen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Lili</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2832358/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Xiaolin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Jiamao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Shanghai Institute of Microsystem and Information Technology, Chinese Academy of Sciences</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Information Science and Technology, ShanghaiTech University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>University of Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Alois C. Knoll, Technical University of Munich, Germany</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Xiongkuo Min, Shanghai Jiao Tong University, China</p>
<p>Hao Chen, Chinese Academy of Sciences (CAS), China</p>
<p>Shenghua Gao, The University of Hong Kong, Hong Kong SAR, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Dongchen Zhu <email>dchzhu&#x00040;mail.sim.ac.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1489021</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Xu, Liu, Zhu, Chen, Zhang and Li.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Xu, Liu, Zhu, Chen, Zhang and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Panoptic segmentation plays a crucial role in enabling robots to comprehend their surroundings, providing fine-grained scene understanding information for robots&#x00027; intelligent tasks. Although existing methods have made some progress, they are prone to fail in areas with weak textures, small objects, etc. Inspired by biological vision research, we propose a cascaded contour-enhanced panoptic segmentation network called CCPSNet, attempting to enhance the discriminability of instances through structural knowledge. To acquire the scene structure, a cascade contour detection stream is designed, which extracts comprehensive scene contours using channel regulation structural perception module and coarse-to-fine cascade strategy. Furthermore, the contour-guided multi-scale feature enhancement stream is developed to boost the discrimination ability for small objects and weak textures. The stream integrates contour information and multi-scale context features through structural-aware feature modulation module and inverse aggregation technique. Experimental results show that our method improves accuracy on the Cityscapes (61.2 PQ) and COCO (43.5 PQ) datasets while also demonstrating robustness in challenging simulated real-world complex scenarios faced by robots, such as dirty cameras and rainy conditions. The proposed network promises to help the robot perceive the real scene. In future work, an unsupervised training strategy for the network could be explored to reduce the training cost.</p></abstract>
<kwd-group>
<kwd>robot vision</kwd>
<kwd>panoptic segmentation</kwd>
<kwd>panoptic contour detection</kwd>
<kwd>structure perception</kwd>
<kwd>cascade</kwd>
<kwd>feature enhancement</kwd>
<kwd>visual pathway</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="5"/>
<equation-count count="9"/>
<ref-count count="52"/>
<page-count count="12"/>
<word-count count="7401"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>In recent years, camera-based perception systems have been widely used in various kinds of robots, which brings the need for image-based scene understanding algorithms. For robots, it is not only necessary to recognize the semantic information in the scene, but also to distinguish different instances, which is of great significance for robots to navigate, interact and execute tasks in complex environments. Specifically, robots need the help of computer vision techniques to identify obstacles in the scene, identify users, find targets, etc.</p>
<p>In order to meet this demand, semantic segmentation (Zhang et al., <xref ref-type="bibr" rid="B48">2022</xref>; Ye et al., <xref ref-type="bibr" rid="B44">2023</xref>; Zhang et al., <xref ref-type="bibr" rid="B49">2023</xref>) and object detection (Liu and Stathaki, <xref ref-type="bibr" rid="B27">2018</xref>; He et al., <xref ref-type="bibr" rid="B13">2017</xref>) tasks have been proposed in the field of computer vision, which are used to identify the semantic information of the pixels in the scene and distinguish the information of different instances in the image respectively. Until Kirillov et al. (<xref ref-type="bibr" rid="B18">2019a</xref>) proposed the panoptic segmentation task, which unifies the above two tasks. This new task is dedicated to identifying each pixel&#x00027;s semantic and instance ID in the input image. This task has gained significant attention in the field of scene understanding due to its precise definition. It is a valuable tool for autonomous driving and industrial robotics applications. There are currently three categories of deep learning-based panoptic segmentation methods based on the instance mask generation approach: top-down (Mask R-CNN based), bottom-up (DeepLab based), and transformer-based (DETR based). These methods have shown promising progress on the datasets. However, misclassification of textureless regions and missing detection of small targets remain to be solved. For example, the white truck was wrongly identified as a building because it lacks texture and has the same color as the building behind it. Additionally, the person in the distance was not detected by the current algorithm due to their small size, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Visualization examples of misclassification and missing detection. <bold>Left</bold>: input image; <bold>middle</bold>: UPSNet results, misclassification of the track and missing detection of the human; <bold>right</bold>: our results.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1489021-g0001.tif"/>
</fig>
<p>The lack of contour perception may cause the problems above, according to the perceptual theory of biological vision. Research (Zhou et al., <xref ref-type="bibr" rid="B52">2000</xref>) indicated that contour detection plays a crucial role in processing scene data in a monkey&#x00027;s visual cortex. The studies suggest that about 18% of the cells in area V1 and over 50% in V2 and V4 on the cerebral cortex are dedicated to processing contour-related information. It is well verified by the previous semantic segmentation that introduced edge detection as an auxiliary task. Gated-SCNN (Takikawa et al., <xref ref-type="bibr" rid="B34">2019</xref>), DecoupleSegNets (Li et al., <xref ref-type="bibr" rid="B22">2020</xref>), and RPCNet (Zhen et al., <xref ref-type="bibr" rid="B51">2020</xref>) demonstrated the importance of contours in scene recognition by introducing semantic edges to improve semantic segmentation performance. In the panoptic segmentation task, some previous studies (Xu et al., <xref ref-type="bibr" rid="B41">2021</xref>; Chang et al., <xref ref-type="bibr" rid="B3">2023</xref>) have explored contour detection as a separate component in their work. CAPSNet (Xu et al., <xref ref-type="bibr" rid="B41">2021</xref>) was the first work that introduced a contour branch to guide feature extraction and explicit contour supervision on the result of panoptic segmentation to improve the network&#x00027;s understanding of the structure. However, this method did not incorporate contour information into the panoptic segmentation process. SE-PSNet (Chang et al., <xref ref-type="bibr" rid="B3">2023</xref>) adopted a similar structure and used contour as auxiliary information to enhance instance segmentation, but did not contribute to semantic segmentation.</p>
<p>In this study, we introduce the Cascade Contour-enhanced Panoptic Segmentation Network (CCPSNet), a novel approach designed to fully embrace structural knowledge to improve the detection of small targets and the semantic recognition of weak texture areas. Our method employs a cascading strategy to adaptively refine multi-scale structural contour details, thereby facilitating more precise detection of contours in areas lacking texture or containing small objects. Furthermore, we present a contour-guided multi-scale feature enhancement stream that integrates panoptic contours and multi-scale features to refine segmentation features and calibrates the perceptual field using structural-aware feature modulation module(SFMM), thereby enhancing segmentation accuracy. The key contributions of our work are as follows:</p>
<list list-type="bullet">
<list-item><p>We propose a cascade contour-enhanced panoptic segmentation network, which effectively delves into the comprehensive structural knowledge using panoptic contour detection and corresponding features, thereby improving the robot vision&#x00027;s perception ability in challenging complex areas.</p></list-item>
<list-item><p>We develop a cascaded contour detection stream for the panoptic segmentation network, which aims to extract scene structural information by a feature channel regulation module and cascade strategy.</p></list-item>
<list-item><p>We design a contour-guided multi-scale feature enhancement stream that incorporates contour information and contextual features to enhance the feature learning of areas with small objects and weak textures.</p></list-item>
<list-item><p>Extensive experiments on the Cityscapes and COCO datasets substantiate the robustness and superiority of our proposed network compared to existing methods.</p></list-item>
</list>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>In this section, we review the progress of research on deep learning-based panoptic segmentation algorithms and contour detection in deep learning.</p>
<sec>
<title>2.1 Deep learning-based panoptic segmentation</title>
<sec>
<title>2.1.1 Top-down</title>
<p>Due to the exceptional performance of Mask R-CNN (He et al., <xref ref-type="bibr" rid="B13">2017</xref>) on the instance segmentation task, this type of approach combines semantic segmentation branches with instance segmentation outcomes to generate panoptic segmentation results. The Panoptic-FPN was proposed by Kirillov, which utilized semantic segmentation with a shared Feature Pyramid Network backbone. Since then, many studies (Chen Y. et al., <xref ref-type="bibr" rid="B6">2020</xref>; Li et al., <xref ref-type="bibr" rid="B23">2019</xref>; Liu et al., <xref ref-type="bibr" rid="B26">2019</xref>; Xiong et al., <xref ref-type="bibr" rid="B40">2019</xref>) have expanded upon this approach. UPSNet (Xiong et al., <xref ref-type="bibr" rid="B40">2019</xref>) introduced a parameter-free panoptic head that predicts the final panoptic segmentation via pixel-wise classification, the number of classes per image of which could vary. BANet (Chen Y. et al., <xref ref-type="bibr" rid="B6">2020</xref>) exploits the complementary relationship between semantics and instances to design Semantic-to-Instance module and Instance-to-semantic module to improve the performance. EfficientPS (Mohan and Valada, <xref ref-type="bibr" rid="B31">2021</xref>) is currently the most effective technique in top-down methods. It involves creating a new backbone network and implementing a 2-way FPN while maintaining the core structure of the Mask R-CNN component.</p>
<p>Based on the above studies, CAPSNet (Xu et al., <xref ref-type="bibr" rid="B41">2021</xref>) pioneered the idea of enhancing the network&#x00027;s ability to perceive structures by introducing panoptic contour-aware branches. Then SE-PSNet (Chang et al., <xref ref-type="bibr" rid="B3">2023</xref>) introduced the contour-based enhancement features into different predicted heads. While the previous methods use contour perception to aid in the understanding of structural information within an image, the design of this approach is relatively simple and may struggle with small targets. To address this issue, we introduce a new cascaded panoptic contour detection head to improve detail awareness and a contour attention feature enhancement module to enhance feature expression. These improvements should enhance the overall performance of the system.</p>
</sec>
<sec>
<title>2.1.2 Bottom-up</title>
<p>In contrast to the above approaches that use instance partitioning as the core of the network, many studies (Chen et al., <xref ref-type="bibr" rid="B4">2017</xref>; Yang et al., <xref ref-type="bibr" rid="B43">2019</xref>; Cheng et al., <xref ref-type="bibr" rid="B7">2020</xref>; Gao et al., <xref ref-type="bibr" rid="B11">2019</xref>; Wang H. et al., <xref ref-type="bibr" rid="B37">2020</xref>; Sun et al., <xref ref-type="bibr" rid="B33">2023</xref>) that focus more on semantic segmentation and cluster the results to generate instance segmentation results. They adopt models such as DeepLab (Chen et al., <xref ref-type="bibr" rid="B4">2017</xref>), a semantic segmentation model that uses an encoder-decoder architecture with an atrous convolution structure as the backbone to generate semantic segmentation results while generating instance segmentation results through a bottom-up approach. Panoptic-DeepLab (Cheng et al., <xref ref-type="bibr" rid="B7">2020</xref>) adopts the dual-ASPP and dual-decoder structures specific to semantic and instance segmentation, respectively. Deeper-Lab (Yang et al., <xref ref-type="bibr" rid="B43">2019</xref>) uses bounding box and center point to generate the instance mask and SSAP (Gao et al., <xref ref-type="bibr" rid="B11">2019</xref>) proposes a pixel-pair affinity pyramid to predict the instance by computing the probability that two pixels belong to the same instance.</p>
</sec>
<sec>
<title>2.1.3 Transformer based</title>
<p>The DEtection TRansformer (DETR) has been proposed by Carion et al. (<xref ref-type="bibr" rid="B2">2020</xref>) as a successful application of the Transformer method, commonly used in NLP, for image detection tasks. Many networks (Wang et al., <xref ref-type="bibr" rid="B36">2021</xref>; Yu et al., <xref ref-type="bibr" rid="B45">2022a</xref>,<xref ref-type="bibr" rid="B46">b</xref>) have subsequently emerged that employ the self-attention module. Max-DeepLab (Wang et al., <xref ref-type="bibr" rid="B36">2021</xref>) introduces mask transformer to predict class-labeled masks directly, while training with panoptic quality inspired loss via bipartite matching to improve Axial-DeepLab (Wang H. et al., <xref ref-type="bibr" rid="B37">2020</xref>)&#x00027;s performance on highly deformable objects, or nearby objects with close centers. Building on this work, CMT-DeepLab (Yu et al., <xref ref-type="bibr" rid="B45">2022a</xref>) composes the process of assigning pixels to the clusters by feature affinity and updating the cluster centers and pixel features as Clustering Mask Transformer. kMaX-DeepLab (Yu et al., <xref ref-type="bibr" rid="B46">2022b</xref>) draws on the k-means clustering algorithm and redesigns the cross-attention mechanism by introducing the relationship between pixels and object queries. The above model greatly simplifies the process of panoptic segmentation and has more powerful feature learning capability to improve the performance effectively. However, they require huge computility is unsatisfactory.</p>
</sec>
</sec>
<sec>
<title>2.2 Contour detection in deep learning</title>
<p>The edge detection task is a fundamental task in computer vision, and this task has also seen new advances through deep learning this year. Among them, HED (Xie and Tu, <xref ref-type="bibr" rid="B39">2015</xref>) improves the performance by using a pattern of fusion of multiple layers. Meanwhile, edges also play an important role in the segmentation task. Nvidia (Takikawa et al., <xref ref-type="bibr" rid="B34">2019</xref>) proposed to assist segmentation through contours. CAPSNet (Xu et al., <xref ref-type="bibr" rid="B41">2021</xref>) is the first model that proposes to introduce panoptic segmentation contours into the panoptic task. SE-PSNet (Chang et al., <xref ref-type="bibr" rid="B3">2023</xref>) assists panoptic segmentation according to semantic contours and instance contours, respectively.</p>
<p>All of the above methods use contouring as an auxiliary task, and for the first time, our model incorporates panoptic segmentation contouring results into the prediction process.</p>
</sec>
<sec>
<title>2.3 Attention model in deep learning</title>
<p>Many previous works have demonstrated the outstanding performance of attention mechanisms on various tasks such as object detection (Alazeb et al., <xref ref-type="bibr" rid="B1">2024</xref>), autonomous driving (Yang et al., <xref ref-type="bibr" rid="B42">2023</xref>), saliency prediction (Min et al., <xref ref-type="bibr" rid="B30">2020</xref>), and fixation prediction (Min et al., <xref ref-type="bibr" rid="B29">2016</xref>). Min et al. (<xref ref-type="bibr" rid="B29">2016</xref>) utilizes canonical correlation analysis to identify the most relevant audio features. It construct visual attention models through spatial attention and temporal attention to predict fixation points. The moving sound target is located using cross-modal kernel canonical correlation analysis. Min et al. (<xref ref-type="bibr" rid="B30">2020</xref>) introduces a two-stage adaptive audiovisual saliency fusion method to complete the saliency prediction. Axial-DeepLab (Wang H. et al., <xref ref-type="bibr" rid="B37">2020</xref>) is a fully attentional network with novel position-sensitive axial-attention layers that combine self-attention for non-local interactions with positional sensitivity. The deletion of the object detection branch leads to those methods being more efficient rather than more effective.</p>
<p>In this paper, we use the attention mechanism to enhance structure perception. The expression of the contour on the feature channel is guided by the attention in the cascaded contour detection stream, and the panoptic contour is used as the input to generate spatial attention to assist the final panoptic segmentation in the contour-guided multi-scale feature enhancement stream.</p>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methodology</title>
<p>In this section, we provide a detailed overview of our proposed network, illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. The network follows a top-down, which employs a shared ResNet backbone with a Feature Pyramid Network (FPN). Our network has four major components: (1) Cascade contour detection stream. (2) Contour-guided multi-scale feature enhancement stream to enhance feature expression on the backbone. (3) The instance segmentation head provides the instance segmentation predictions. (4) The semantic segmentation head predicts semantic results. This section provides the details of those.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Illustration of the proposed CCPSNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1489021-g0002.tif"/>
</fig>
<sec>
<title>3.1 Cascade contour detection stream</title>
<p>Inspired by Gated-SCNN (Takikawa et al., <xref ref-type="bibr" rid="B34">2019</xref>) and RPCNet (Zhen et al., <xref ref-type="bibr" rid="B51">2020</xref>) that introduce the edge detection to aid in semantic segmentation, contours are critical clues in segmentation task. In order to outline every object or background in the scene, panoptic segmentation contour is an amalgamation of semantic segmentation contour and instance segmentation contour. Particularly, for an image <italic>I</italic>, it&#x00027;s panoptic segmentation contour label <italic>C</italic><sub><italic>I</italic></sub> &#x0003D; <italic>C</italic><sub><italic>s</italic></sub>&#x0222A;<italic>C</italic><sub><italic>t</italic></sub>. Here, <italic>C</italic><sub><italic>s</italic></sub> is the semantic contour for stuff categories and <italic>C</italic><sub><italic>t</italic></sub> is the instance contour for <italic>things</italic> label. In terms of the truth result of the panoptic segmentation ground truth <italic>P</italic>, specifically for a pixel <italic>P</italic><sub><italic>p</italic></sub>, whenever any of the 8 pixels surrounding this pixel point has a different semantic or instance ID with it, we consider it to be a panoptic contour pixel. Since the results are colored by category and instance, different categories or different objects id in the same category will be colored differently. Thus Laplace convolution can be used to obtain a panoptic segmentation contour.</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mtext>&#x02003;</mml:mtext><mml:mo>&#x02200;</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>P</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>s</mml:mi><mml:mo>.</mml:mo><mml:mi>t</mml:mi><mml:mo>.</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>u</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02260;</mml:mo><mml:mn>0</mml:mn><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In order to get the contour, we design a cascade contour detection stream. The structure is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. We introduce the features<italic>P</italic><sub>2</sub>&#x02212;<italic>P</italic><sub>5</sub> obtained from FPN into the channel regulation structural perception module (CRSPM) to obtain the contour features of the corresponding layers. To begin with, we apply a 3 &#x000D7; 3 convolution to convert the features into contour features. Subsequently, considering the variability of the activation patterns of different convolution kernels, we generate weights on channels to regulate feature volume, which aims to enhance structural information and suppress negative impact information. This module is primarily built through global average pooling and fully connected layers, which can be represented by the following formula:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02297;</mml:mo><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>G</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In which, <italic>f</italic>() represents the two 3 &#x000D7; 3 convolution layers, <italic>GAP</italic>() represents the global average pooling, and <italic>g</italic>() is the 1 &#x000D7; 1 convolution. By this way, we retain the ability to perceive the texture while picking out the channels that are responsive to the contours. In order to maintain the overall information on the large scale and avoid the misclassification caused by the texture difference inside the structure, the large-scale and small-scale features are fused by concatenate operation. From <italic>P</italic><sub>5</sub> to <italic>P</italic><sub>2</sub>, we apply coarse-to-fine cascade aggregation to get the contour feature and predict the panoptic segmentation contour.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The structure of cascade contour detection stream.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1489021-g0003.tif"/>
</fig>
<p>Due to the distribution imbalance of the number of contour and non-contour pixels, we adopt the class-balancing cross-entropy loss function <italic>L</italic><sub><italic>c</italic></sub> following the HED (Xie and Tu, <xref ref-type="bibr" rid="B39">2015</xref>) as the contour loss. <xref ref-type="disp-formula" rid="E1">Equation 1</xref> provides its formalization. In which &#x003B1; is the class-balancing weight on a per-pixel term basis. <italic>C</italic> denotes the ground truth of panoptic contour, and &#x00108; denotes the predicted panoptic contour. <italic>C</italic><sub>&#x02212;</sub> denotes the contour ground truth label sets.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mi>C</mml:mi><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mover accent='true'><mml:mi>C</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mover accent='true'><mml:mi>C</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>;</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>C</mml:mi><mml:mo>&#x02212;</mml:mo></mml:msub></mml:mrow><mml:mi>C</mml:mi></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>
</sec>
<sec>
<title>3.2 Contour-guided multi-scale feature enhancement stream</title>
<p>In the process of extracting features from the backbone, a lot of detailed information is lost due to multiple down-sampling and pooling operations, which leads to the mission detection of small objects. To address this issue, we proposed structural-aware feature modulation module (SFMM) and inverse aggregation operation, which augments the features by taking the panoptic contours to generate attention at the spatial scale and fuses with features of different scales. This module helps incorporate structured information into features at all scales to aid in learning small object features. Its structure is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. Here, we give a detailed formulation for this process.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The structure of contour-guided multi-scale feature enhancement stream.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1489021-g0004.tif"/>
</fig>
<p>Given an input feature map <inline-formula><mml:math id="M4"><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> from the <italic>i</italic>-th scale FPN branch. The contour was resized to same scale with this feature map <inline-formula><mml:math id="M5"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>. The attention can be formulated as follows:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02297;</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>f</italic><sub><italic>i</italic><sub><italic>j</italic></sub></sub>(&#x000B7;, &#x000B7;) denotes an atrous convolution function, &#x02297; represents element-wise multiplication, &#x003C3; means a 1 &#x000D7; 1 convolution, and <italic>sigmoid</italic> indicates the Sigmoid activation function, <inline-formula><mml:math id="M7"><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the attention generate from the contour. In this process, the atrous convolutions <italic>f</italic> are employed to mine spatial information from contour. During the experiments, in our experience, the kernel sizes of both <italic>f</italic><sub><italic>i</italic><sub>1</sub></sub> and <italic>f</italic><sub><italic>i</italic><sub>2</sub></sub> are set to 15, with a dilation rate of 3, and they do not share weights.</p>
<p>To emphasize the feature of object, we formulate the attention weighted map <inline-formula><mml:math id="M8"><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> as 1&#x02212;<italic>Att</italic><sub><italic>i</italic></sub>. Then the enhanced feature map <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> can be presented as:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02297;</mml:mo><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02295;</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x02295; means element-wise sum.</p>
<p>Inspired by Tan et al. (<xref ref-type="bibr" rid="B35">2020</xref>), an inverse aggregation method is designed to utilize features in low levels assistant for large instance identification. We designed the structure to allow low-level features to provide detailed information for nearby high-level features, which implemented by resizing the low-level feature map <inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> to the same scale as near high-level feature map <inline-formula><mml:math id="M12"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> and employing the element-wise sum. The <italic>i</italic>-th feature <inline-formula><mml:math id="M13"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> for segmentation and detection can be generated as:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02295;</mml:mo><mml:mi>&#x003B4;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B4; means down-sampling. It is worth noting that since <inline-formula><mml:math id="M15"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> has no lower-level features, in practice <inline-formula><mml:math id="M16"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M17"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are the same. At this point, enhancement of features based on contours is complete.</p>
</sec>
<sec>
<title>3.3 Instance segmentation head</title>
<p>Following the Mask R-CNN (He et al., <xref ref-type="bibr" rid="B13">2017</xref>), our instance segmentation head produces bounding box regression, classification, and segmentation mask from <inline-formula><mml:math id="M18"><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula>. The purpose of this head is to provide pixel-level annotations for each instance. The loss function <italic>L</italic><sub><italic>ins</italic></sub> is defined as follows:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>b</mml:mi><mml:mi>o</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>L</italic><sub><italic>cls</italic></sub> is the is the classification loss, <italic>L</italic><sub><italic>bbox</italic></sub> is the object bounding-box regression loss, and <italic>L</italic><sub><italic>mask</italic></sub> is the average binary cross-entropy loss for mask prediction.</p>
</sec>
<sec>
<title>3.4 Semantic segmentation head</title>
<p>For semantic segmentation head, we stack two 3 &#x000D7; 3 deformable convolution layers following SFMM features. Similar to PSPNet (Zhao et al., <xref ref-type="bibr" rid="B50">2017</xref>), we first scale the features of different sizes to the same scale to achieve fusion of different granularity. Then, the semantic presentations are obtained by concatenating these scaled features. For this head, we choose the standard cross entropy in semantic segmentation as the loss function, denoted as <italic>L</italic><sub><italic>seg</italic></sub>.</p>
<p>During training, the total loss <italic>L</italic><sub><italic>total</italic></sub> is formulated as:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments</title>
<p>In this section, our CCPSNet is evaluated on Cityscapes (Cordts et al., <xref ref-type="bibr" rid="B9">2016</xref>) and Microsoft COCO (Lin et al., <xref ref-type="bibr" rid="B25">2014</xref>) datasets. We present the experimental results on these datasets and compare with them the state-of-the-art models based on Mask R-CNN. The ablation studies and robustness analysis are presented at last.</p>
<sec>
<title>4.1 Datasets and metrics</title>
<sec>
<title>4.1.1 Cityscapes</title>
<p>This dataset focuses on understanding urban streets scenes. It is composed of 2,975 training images, 500 validation images, and 1,525 test images. All these 5,000 images are with fine annotations. There are another 20,000 images with coarse annotations, which are not utilized in our experiment. This dataset has a total of 19 categories, of which 8 are things and the remaining 11 are stuff. In this dataset, we use 4 RTX 2080Ti GPUs to train our model, and trained in a batch size of 1 per GPU, learning rate of 0.02 and weight decay of 1<italic>e</italic><sup>&#x02212;4</sup> for 48,000 steps in total and decay the learning rate by a factor of 0.1 at step 36,000.</p>
</sec>
<sec>
<title>4.1.2 COCO</title>
<p>This dataset contains a large number of natural images, comprising both indoor and outdoor scenes. The distribution between images is also inconsistent, making it challenging for algorithms to learn good results. It contains 140,000 images with 115,000 training images, 5,000 validation images, 20,000 test-dev images, and 20,000 test images. There are 80 thing categories and 53 stuff categories. We only rely on the train set with no extra data, presenting the results on the validation set for comparison. In this dataset, we use 8 RTX 2080Ti GPUs to train our model, and trained in a batch size of 1 per GPU, learning rate of 0.01 and weight decay of 1<italic>e</italic><sup>&#x02212;4</sup> for 48,000 steps in total and decay the learning rate by a factor of 0.1 at step 240,000 and 32,000.</p>
</sec>
<sec>
<title>4.1.3 Evaluation metrics</title>
<p>Following Kirillov et al. (<xref ref-type="bibr" rid="B19">2019b</xref>), the Panoptic Quality (PQ) is adopted as evaluation metrics:</p>
<disp-formula id="E9"><mml:math id="M21"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mi>Q</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mstyle displaystyle="true"><mml:munder accentunder="false"><mml:mrow><mml:mfrac><mml:mrow><mml:mstyle displaystyle="false"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mrow><mml:mo>&#x0FE38;</mml:mo></mml:munder></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>Q</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mo>&#x000D7;</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mstyle displaystyle="true"><mml:munder accentunder="false"><mml:mrow><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mo>|</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mo>|</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mrow><mml:mo>&#x0FE38;</mml:mo></mml:munder></mml:mstyle></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mi>Q</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>According to the formula, PQ is determined by multiplying SQ and RQ, combining the evaluation of semantic segmentation and instance segmentation. In the formula, <italic>IoU</italic>(<italic>p, g</italic>) means the intersection-over-union between predicted object <italic>p</italic> and ground truth <italic>g</italic>. <italic>TP</italic> (True Positives) represents matched pairs of segments, and <italic>FP</italic> (False Positive) means unmatched pairs of segments, and <italic>FN</italic> (False Negatives) means unmatched ground truth segments. When the value of <italic>IoU</italic> is &#x0003E;0.5, it is considered a positive match. Note that <italic>PQ</italic>, <italic>PQ</italic><sup><italic>Th</italic></sup>, and <italic>PQ</italic><sup><italic>St</italic></sup> refer to the <italic>PQ</italic> values averaged across all classes, thing classes and stuff classes.</p>
</sec>
</sec>
<sec>
<title>4.2 Comparsion to state-of-the-art</title>
<p>We compare our proposed network with other state-of-the-art methods on Cityscapes (Cordts et al., <xref ref-type="bibr" rid="B9">2016</xref>) val set and MS-COCO (Lin et al., <xref ref-type="bibr" rid="B25">2014</xref>).</p>
<sec>
<title>4.2.1 Cityscapes</title>
<p>As shown in <xref ref-type="table" rid="T1">Table 1</xref>, the proposed CCPSNet of ResNet-50 is used to realize 60.5% and CCPSNet of ResNet-101 is used to realize 61.2% which is improved compared with similar methods with same backbone. <xref ref-type="fig" rid="F5">Figure 5</xref> presents some visual examples of our algorithm on Cityscapes. The first row shows that the problem of small target objects, such as cyclists, which is difficult to distinguish due to lighting problems is alleviated by the feature cascade fusion module proposed by CCPSNet. The second and third rows show that when there is occlusion between different objects, contour perception in CCPSNet can solve this problem very well. The fourth row shows that through contour perception and contour feature enhancement, CCPSNet can effectively detect ambiguous scenes such as the left side of the vehicle painted with a face pattern and the right side of the street hanging a row of clothes.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparsion with other methods on Cityscapes val sets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold><italic>PQ</italic></bold></th>
<th valign="top" align="center"><bold><italic>PQ</italic><sup><italic>Th</italic></sup></bold></th>
<th valign="top" align="center"><bold><italic>PQ</italic><sup><italic>St</italic></sup></bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="4"><bold>Backbone: ResNet-50 (He et al.</bold>, <xref ref-type="bibr" rid="B14"><bold>2016</bold></xref><bold>)</bold></td>
</tr>
<tr>
<td valign="top" align="left">EfficientPS (Mohan and Valada, <xref ref-type="bibr" rid="B31">2021</xref>)</td>
<td valign="top" align="center">60.3</td>
<td valign="top" align="center">55.3</td>
<td valign="top" align="center">63.9</td>
</tr>
<tr>
<td valign="top" align="left">Panoptic-FPN (Kirillov et al., <xref ref-type="bibr" rid="B18">2019a</xref>)</td>
<td valign="top" align="center">57.7</td>
<td valign="top" align="center">51.6</td>
<td valign="top" align="center">62.2</td>
</tr>
<tr>
<td valign="top" align="left">UPSNet (Xiong et al., <xref ref-type="bibr" rid="B40">2019</xref>)</td>
<td valign="top" align="center">59.1</td>
<td valign="top" align="center">54.1</td>
<td valign="top" align="center">62.7</td>
</tr>
<tr>
<td valign="top" align="left">AUNet (Li et al., <xref ref-type="bibr" rid="B23">2019</xref>)</td>
<td valign="top" align="center">56.4</td>
<td valign="top" align="center">52.7</td>
<td valign="top" align="center">59.0</td>
</tr>
<tr>
<td valign="top" align="left">CAPSNet (Xu et al., <xref ref-type="bibr" rid="B41">2021</xref>)</td>
<td valign="top" align="center">60.0</td>
<td valign="top" align="center">55.7</td>
<td valign="top" align="center">63.1</td>
</tr>
<tr>
<td valign="top" align="left">YOSO (Hu et al., <xref ref-type="bibr" rid="B16">2023</xref>)</td>
<td valign="top" align="center">59.7</td>
<td valign="top" align="center">51.0</td>
<td valign="top" align="center"><bold>66.1</bold></td>
</tr>
<tr>
<td valign="top" align="left">LPSNet (Hong et al., <xref ref-type="bibr" rid="B15">2021</xref>)</td>
<td valign="top" align="center">59.7</td>
<td valign="top" align="center">54.0</td>
<td valign="top" align="center">63.9</td>
</tr>
<tr>
<td valign="top" align="left">SE-PSNet (Chang et al., <xref ref-type="bibr" rid="B3">2023</xref>)</td>
<td valign="top" align="center">60.0</td>
<td valign="top" align="center">55.9</td>
<td valign="top" align="center">62.9</td>
</tr>
<tr>
<td valign="top" align="left">CCPSNet (Ours)</td>
<td valign="top" align="center"><bold>60.5</bold></td>
<td valign="top" align="center"><bold>56.9</bold></td>
<td valign="top" align="center">63.1</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="4"><bold>Backbone: ResNet-101 (He et al.</bold>, <xref ref-type="bibr" rid="B14"><bold>2016</bold></xref><bold>)</bold></td>
</tr>
<tr>
<td valign="top" align="left">EfficientPS (Mohan and Valada, <xref ref-type="bibr" rid="B31">2021</xref>)</td>
<td valign="top" align="center">61.1</td>
<td valign="top" align="center">56.5</td>
<td valign="top" align="center"><bold>64.2</bold></td>
</tr>
<tr>
<td valign="top" align="left">Panoptic-FPN (Kirillov et al., <xref ref-type="bibr" rid="B18">2019a</xref>)</td>
<td valign="top" align="center">58.1</td>
<td valign="top" align="center">52.0</td>
<td valign="top" align="center">62.5</td>
</tr>
<tr>
<td valign="top" align="left">AUNet (Li et al., <xref ref-type="bibr" rid="B23">2019</xref>)</td>
<td valign="top" align="center">59.0</td>
<td valign="top" align="center">54.8</td>
<td valign="top" align="center">62.1</td>
</tr>
<tr>
<td valign="top" align="left">AdaptIS (Sofiiuk et al., <xref ref-type="bibr" rid="B32">2019</xref>)</td>
<td valign="top" align="center">60.6</td>
<td valign="top" align="center"><bold>57.5</bold></td>
<td valign="top" align="center">62.6</td>
</tr>
<tr>
<td valign="top" align="left">CCPSNet (Ours)</td>
<td valign="top" align="center"><bold>61.2</bold></td>
<td valign="top" align="center">57.1</td>
<td valign="top" align="center">64.1</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold value means the best result in the column.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Visual examples of panoptic segmentation on Cityscapes. From left to right are input images, predicted results from UPSNet, CAPSNet, CCPSNet (ours), and ground truth.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1489021-g0005.tif"/>
</fig>
<p>We evaluate the segmentation results on specific categories in Cityscapes, where the object size is smaller than 32 pixels &#x000D7; 32 pixels. The results based on ResNet-50 are reported in <xref ref-type="table" rid="T2">Table 2</xref>. It can be observed that our proposed algorithm indeed improves the performance for small objects, thanks to contour-guided multi-scale feature enhancement stream.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Accuracy of small object on CityScapes val set.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold><italic>PQ</italic></bold></th>
<th valign="top" align="center"><bold><italic>SQ</italic></bold></th>
<th valign="top" align="center"><bold><italic>RQ</italic></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">UPSNet (Xiong et al., <xref ref-type="bibr" rid="B40">2019</xref>)</td>
<td valign="top" align="center">50.26</td>
<td valign="top" align="center">71.70</td>
<td valign="top" align="center">70.13</td>
</tr>
<tr>
<td valign="top" align="left">CAPSNet (Xu et al., <xref ref-type="bibr" rid="B41">2021</xref>)</td>
<td valign="top" align="center">51.01</td>
<td valign="top" align="center"><bold>72.70</bold></td>
<td valign="top" align="center">70.19</td>
</tr>
<tr>
<td valign="top" align="left">CCPSNet (Ours)</td>
<td valign="top" align="center"><bold>51.26</bold></td>
<td valign="top" align="center">72.53</td>
<td valign="top" align="center"><bold>70.70</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold value means the best result in the column.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.2.2 COCO</title>
<p>In addition to verification in outdoor driving scenarios, we further demonstrated the universality of our method on the COCO dataset, which includes various indoor and outdoor scenes. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, we compare with similar methods with same backbone. The proposed CCPSNet achieves the PQ 43.2% and the PQ 43.5% with backbone ResNet-101.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Comparsion with other methods on COCO val sets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><italic>PQ</italic></th>
<th valign="top" align="center"><italic>PQ</italic><sup><italic>Th</italic></sup></th>
<th valign="top" align="center"><italic>PQ</italic><sup><italic>St</italic></sup></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="4"><bold>Backbone: ResNet-50 (He et al.</bold>, <xref ref-type="bibr" rid="B14"><bold>2016</bold></xref><bold>)</bold></td>
</tr>
<tr>
<td valign="top" align="left">UPSNet (Xiong et al., <xref ref-type="bibr" rid="B40">2019</xref>)</td>
<td valign="top" align="center">42.5</td>
<td valign="top" align="center">48.5</td>
<td valign="top" align="center">33.4</td>
</tr>
<tr>
<td valign="top" align="left">TASCNet (Li et al., <xref ref-type="bibr" rid="B21">2018</xref>)</td>
<td valign="top" align="center">40.7</td>
<td valign="top" align="center">47.0</td>
<td valign="top" align="center">31.0</td>
</tr>
<tr>
<td valign="top" align="left">SpatialFlow (Chen Q. et al., <xref ref-type="bibr" rid="B5">2020</xref>)</td>
<td valign="top" align="center">42.9</td>
<td valign="top" align="center">49.5</td>
<td valign="top" align="center">33.0</td>
</tr>
<tr>
<td valign="top" align="left">Panoptic FPN (Kirillov et al., <xref ref-type="bibr" rid="B18">2019a</xref>)</td>
<td valign="top" align="center">39.0</td>
<td valign="top" align="center">45.9</td>
<td valign="top" align="center">28.7</td>
</tr>
<tr>
<td valign="top" align="left">OANet (Liu et al., <xref ref-type="bibr" rid="B26">2019</xref>)</td>
<td valign="top" align="center">41.3</td>
<td valign="top" align="center"><bold>50.4</bold></td>
<td valign="top" align="center">27.7</td>
</tr>
<tr>
<td valign="top" align="left">JSIS-Net (De Geus et al., <xref ref-type="bibr" rid="B10">2018</xref>)</td>
<td valign="top" align="center">26.9</td>
<td valign="top" align="center">29.3</td>
<td valign="top" align="center">23.3</td>
</tr>
<tr>
<td valign="top" align="left">AUNet (Li et al., <xref ref-type="bibr" rid="B23">2019</xref>)</td>
<td valign="top" align="center">39.6</td>
<td valign="top" align="center">49.1</td>
<td valign="top" align="center">25.2</td>
</tr>
<tr>
<td valign="top" align="left">AdaptIS (Sofiiuk et al., <xref ref-type="bibr" rid="B32">2019</xref>)</td>
<td valign="top" align="center">35.9</td>
<td valign="top" align="center">40.3</td>
<td valign="top" align="center">29.3</td>
</tr>
<tr>
<td valign="top" align="left">CIAE (Gao et al., <xref ref-type="bibr" rid="B12">2021</xref>)</td>
<td valign="top" align="center">40.2</td>
<td valign="top" align="center">45.3</td>
<td valign="top" align="center">32.3</td>
</tr>
<tr>
<td valign="top" align="left">SOLO V2 (Wang X. et al., <xref ref-type="bibr" rid="B38">2020</xref>)</td>
<td valign="top" align="center">42.1</td>
<td valign="top" align="center">49.6</td>
<td valign="top" align="center">30.7</td>
</tr>
<tr>
<td valign="top" align="left">OCFusion (Lazarow et al., <xref ref-type="bibr" rid="B20">2020</xref>)</td>
<td valign="top" align="center">41.3</td>
<td valign="top" align="center">49.4</td>
<td valign="top" align="center">29.0</td>
</tr>
<tr>
<td valign="top" align="left">LPSNet (Hong et al., <xref ref-type="bibr" rid="B15">2021</xref>)</td>
<td valign="top" align="center">39.1</td>
<td valign="top" align="center">43.9</td>
<td valign="top" align="center">30.1</td>
</tr>
<tr>
<td valign="top" align="left">IDNet (Lin et al., <xref ref-type="bibr" rid="B24">2023</xref>)</td>
<td valign="top" align="center">42.1</td>
<td valign="top" align="center">47.5</td>
<td valign="top" align="center">33.9</td>
</tr>
<tr>
<td valign="top" align="left">CCPSNet (Ours)</td>
<td valign="top" align="center"><bold>43.0</bold></td>
<td valign="top" align="center">49.2</td>
<td valign="top" align="center"><bold>33.6</bold></td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="4"><bold>Backbone: ResNet-101 (He et al.</bold>, <xref ref-type="bibr" rid="B14"><bold>2016</bold></xref><bold>)</bold></td>
</tr>
<tr>
<td valign="top" align="left">Panoptic-FPN (Kirillov et al., <xref ref-type="bibr" rid="B18">2019a</xref>)</td>
<td valign="top" align="center">40.3</td>
<td valign="top" align="center">47.5</td>
<td valign="top" align="center">29.5</td>
</tr>
<tr>
<td valign="top" align="left">AdaptIS (Sofiiuk et al., <xref ref-type="bibr" rid="B32">2019</xref>)</td>
<td valign="top" align="center">37.0</td>
<td valign="top" align="center">41.8</td>
<td valign="top" align="center">29.9</td>
</tr>
<tr>
<td valign="top" align="left">OCFusion (Lazarow et al., <xref ref-type="bibr" rid="B20">2020</xref>)</td>
<td valign="top" align="center">43.0</td>
<td valign="top" align="center"><bold>51.1</bold></td>
<td valign="top" align="center">30.7</td>
</tr>
<tr>
<td valign="top" align="left">SSAP (Gao et al., <xref ref-type="bibr" rid="B11">2019</xref>)</td>
<td valign="top" align="center">36.9</td>
<td valign="top" align="center">40.1</td>
<td valign="top" align="center">32.0</td>
</tr>
<tr>
<td valign="top" align="left">CCPSNet (Ours)</td>
<td valign="top" align="center"><bold>43.5</bold></td>
<td valign="top" align="center">49.9</td>
<td valign="top" align="center"><bold>33.8</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold value means the best result in the column.</p>
</table-wrap-foot>
</table-wrap>
<p><xref ref-type="fig" rid="F6">Figure 6</xref> presents some visual examples of our algorithm on MS-COCO. The first row presented is similar to the one in CityScapes in its ability to detect extra-long targets such as trains, and the distant train body is well preserved. The second row is mainly for the detection of bus drivers, and the information of the characters is better retained. The third row focuses on the detection of small targets of bird flocks. Unlike UPSNet, CCPSNet still retains the good individual characteristics of flying birds for a large number of small targets and does not show a large number of block-like structures, and there are relatively independent characteristics among flying birds from the original figure. The fourth row shows the structure of the flush toilet, which is well preserved in CCPSNet, even if the structure of the drain is clearly visible.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Visual examples of panoptic segmentation on COCO. From left to right are input images, predicted results from UPSNet, CCPSNet (ours), and ground truth.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1489021-g0006.tif"/>
</fig>
</sec>
</sec>
<sec>
<title>4.3 Ablation studies</title>
<p>To demonstrate the effectiveness of each component in our network, we conduct related ablation experiments. <xref ref-type="table" rid="T4">Table 4</xref> shows the quantitative ablative analysis, where empty cells mean the corresponding components are not adopted.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Results of ablation experiments on CityScapes val set.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left" colspan="2"><bold>Cascade contour detection stream</bold></th>
<th valign="top" align="center" colspan="2"><bold>Contour-guided multi-scale feature enhancement stream</bold></th>
<th valign="top" align="center"><bold>PQ</bold></th>
<th valign="top" align="center"><bold>PQ<sup><italic>th</italic></sup></bold></th>
<th valign="top" align="center"><bold>PQ<sup><italic>st</italic></sup></bold></th>
<th valign="top" align="center"><bold>SQ</bold></th>
<th valign="top" align="center"><bold>RQ</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td valign="top" align="left"><bold>Cascade structure</bold></td>
<td valign="top" align="center"><bold>CRSPM</bold></td>
<td valign="top" align="center"><bold>SFMM</bold></td>
<td valign="top" align="center"><bold>Inverse aggregation</bold></td>
<td/>
<td/>
<td/>
<td/>
<td/>
</tr>
<tr>
<td/>
<td/>
<td/>
<td/>
<td valign="top" align="center">59.1</td>
<td valign="top" align="center">54.1</td>
<td valign="top" align="center">62.7</td>
<td valign="top" align="center">80.1</td>
<td valign="top" align="center">72.4</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td/>
<td/>
<td/>
<td valign="top" align="center">59.6</td>
<td valign="top" align="center">54.9</td>
<td valign="top" align="center">63.0</td>
<td valign="top" align="center">80.0</td>
<td valign="top" align="center">73.2</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td/>
<td valign="top" align="center">59.8</td>
<td valign="top" align="center">55.2</td>
<td valign="top" align="center"><bold>63.1</bold></td>
<td valign="top" align="center">80.0</td>
<td valign="top" align="center">73.4</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td/>
<td valign="top" align="center">60.1</td>
<td valign="top" align="center">55.9</td>
<td valign="top" align="center"><bold>63.1</bold></td>
<td valign="top" align="center">80.2</td>
<td valign="top" align="center">73.6</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center"><bold>60.5</bold></td>
<td valign="top" align="center"><bold>56.9</bold></td>
<td valign="top" align="center"><bold>63.1</bold></td>
<td valign="top" align="center"><bold>80.3</bold></td>
<td valign="top" align="center"><bold>74.1</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold value means the best result in the column.</p>
</table-wrap-foot>
</table-wrap>
<p>In <xref ref-type="table" rid="T4">Table 4</xref>, the first row shows the results of the baseline without any innovative design. The second row represents the experimental results for the cascade contour stream without CRSPM we designed. This outcome demonstrates that the introduction of contours effectively enhances the scene perception capabilities, particularly in terms of the overall evaluation metric PQ and the object recognition metric RQ. Compared to the second row, the third row exhibits the effectiveness of CRSPM. As can be seen from the results, CRSPM further improves performance. We believe that this performance enhancement is primarily due to the following reasons. The inclusion of the contour detection head can encourage base feature extractor to focus on learning structual features, and channel regulation structural perception module employ a GAP and 1 &#x000D7; 1 convolution to re-weight the channels, selecting those sensitive to contour perception and allowing them to cascade participate in the perception process. We utilize the same global average pooling operation as in Condori and Bruno (<xref ref-type="bibr" rid="B8">2021</xref>) to preserve the texture information. This further indicates that introducing the contour recognition function of the visual cortex of the brain in the task of scene recognition can effectively improve the performance of performance performance of the network. The cascaded panoptic segmentation contour branch proposed by CCPSNet can perceive the contour more finely.</p>
<p>To validate the effectiveness of contour-guided multi-scale feature enhancement stream, we conducted experimental verification of structural-aware feature modulation module(SFMM) and inverse aggregation based on the third-row model. It is worth noting that the introduction of inverse aggregation improves PQ by 1%, indicating that this design can indeed help improve instance detection. It can be seen that the contour-guided multi-scale feature enhancement stream brings a 0.7% performance gain to the overall network metrics, including a 1.7% performance gain to the foreground panoptic segmentation <italic>PQ</italic><sup><italic>th</italic></sup> and a 0.3% gain to the background panoptic segmentation SQ. This is mainly due to the introduction of contour information and the enhancement of the features with more detailed information through bottom-up feature cascading, which also helps semantic segmentation.</p>
</sec>
<sec>
<title>4.4 Robustness analysis</title>
<p>When a robot perceives its environment in the real world, it encounters various types of distortion at each stage of visual signal acquisition, compression and transmission. Additionally, it may face challenges such as rainy days and camera dirt, which can affect image quality and subsequently impact the algorithm&#x00027;s performance. Many studies (Zhai and Min, <xref ref-type="bibr" rid="B47">2020</xref>; Min et al., <xref ref-type="bibr" rid="B28">2024</xref>) have shown that image quality is essential for artificial intelligence, and low-quality input will impact the algorithm&#x00027;s performance. In this regard, we conduct experimental analysis on the robustness of our method to the input image quality. We applied image processing on the CityScapes dataset to verify the robustness of our network. By calling the imgaug (Jung et al., <xref ref-type="bibr" rid="B17">2020</xref>) library functions, rain and noise were incorporated into the images to simulate challenging conditions such as rainy scenes and dirty cameras, which are commonly encountered by robots during operation. We evaluated the model trained on the original Cityscapes dataset directly on new data without additional training to test the algorithm&#x00027;s robustness. The experimental results, presented in <xref ref-type="table" rid="T5">Table 5</xref>, compare the performance of the proposed method and the UPSNet in these scenarios. In the case of simulated rainy days and dirty cameras, our algorithm achieved PQ scores of 50.4% and 39.7%, respectively. The results demonstrate varying degrees of performance degradation compared to the original dataset, but our proposed algorithm continues to outperform in these complex scenarios. This is mainly caused by the different impacts of noise on the image. As shown in <xref ref-type="fig" rid="F7">Figure 7</xref>, it can be seen that the rain image has little change compared with the original image, but the simulated dirty image has a large change. In the top row picture, it is evident that our algorithm still has a strong ability to perceive contours in complex scenes. In the bottom row picture, it is evident that the introduction of contours has improved our performance in dealing with large textureless areas. This demonstrates the promising robustness of the proposed CCPSNet in challenging scenarios faced by robots.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Accuracy on noise CityScapes val set.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Sense</bold></th>
<th valign="top" align="center"><bold>Method</bold></th>
<th valign="top" align="center"><bold><italic>PQ</italic></bold></th>
<th valign="top" align="center"><bold><italic>SQ</italic></bold></th>
<th valign="top" align="center"><bold><italic>RQ</italic></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Dirtiness</td>
<td valign="top" align="center">UPSNet (Xiong et al., <xref ref-type="bibr" rid="B40">2019</xref>)</td>
<td valign="top" align="center">37.9</td>
<td valign="top" align="center">69.9</td>
<td valign="top" align="center">48.8</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">CCPSNet (Ours)</td>
<td valign="top" align="center"><bold>39.7</bold></td>
<td valign="top" align="center"><bold>74.6</bold></td>
<td valign="top" align="center"><bold>51.0</bold></td>
</tr>
<tr>
<td valign="top" align="left">Rainy</td>
<td valign="top" align="center">UPSNet (Xiong et al., <xref ref-type="bibr" rid="B40">2019</xref>)</td>
<td valign="top" align="center">49.5</td>
<td valign="top" align="center">76.9</td>
<td valign="top" align="center">62.5</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">CCPSNet (Ours)</td>
<td valign="top" align="center"><bold>50.4</bold></td>
<td valign="top" align="center"><bold>78.1</bold></td>
<td valign="top" align="center"><bold>62.9</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold value means the best result in the column.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Visualization examples of panoptic segmentation on data augmentation Cityscapes. From left to right are input images, predicted results from UPSNet, CCPSNet (ours), and ground truth.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1489021-g0007.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusions</title>
<p>In this paper, we introduced a novel panoptic segmentation algorithm that relies on panoptic segmentation contour guidance. Our approach proposes a new cascade contour detection stream that coarse-to-fine extracts scene structural information. We also developed a contour-guided multi-scale feature enhancement stream that fully utilizes the extracted contours. In addition, our feature inverse aggregation structure enables a bi-directional flow of features to achieve perceptual enhancement of small objects. Finally, the experimental results on Cityscapes and COCO show that our algorithm is highly competitive with similar algorithms. We verify the robustness of the proposed algorithm in complex environments by simulating rain and camera dirtiness with data augmentation. In future work, we plan to extend this idea to unsupervised segmentation tasks.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>YX: Writing &#x02013; original draft, Validation. RL: Visualization, Writing &#x02013; review &#x00026; editing. DZ: Methodology, Project administration, Writing &#x02013; review &#x00026; editing. LC: Formal analysis, Investigation, Writing &#x02013; review &#x00026; editing. XZ: Writing &#x02013; review &#x00026; editing, Supervision. JL: Funding acquisition, Project administration, Writing &#x02013; review &#x00026; editing, Methodology.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was funded by the National Science and Technology Major Project from Minister of Science and Technology, China (2018AAA0103100), Natural Science Foundation of Shanghai (23ZR1474200), Shanghai Municipal Science and Technology Major Project (ZHANGJIANG LAB) under Grant 2018SHZDZX01, Youth Innovation Promotion Association, Chinese Academy of Sciences (2021233), and the Shanghai Academic Research Leader (22XD1424500).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alazeb</surname> <given-names>A.</given-names></name> <name><surname>Chughtai</surname> <given-names>B. R.</given-names></name> <name><surname>Al Mudawi</surname> <given-names>N.</given-names></name> <name><surname>AlQahtani</surname> <given-names>Y.</given-names></name> <name><surname>Alonazi</surname> <given-names>M.</given-names></name> <name><surname>Aljuaid</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Remote intelligent perception system for multi-object detection</article-title>. <source>Front. Neurorobot</source>. <volume>18</volume>:<fpage>1398703</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2024.1398703</pub-id><pub-id pub-id-type="pmid">38831877</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Carion</surname> <given-names>N.</given-names></name> <name><surname>Massa</surname> <given-names>F.</given-names></name> <name><surname>Synnaeve</surname> <given-names>G.</given-names></name> <name><surname>Usunier</surname> <given-names>N.</given-names></name> <name><surname>Kirillov</surname> <given-names>A.</given-names></name> <name><surname>Zagoruyko</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;End-to-end object detection with transformers,&#x0201D;</article-title> in <source>European conference on computer vision</source> (<publisher-loc>Springer</publisher-loc>), <fpage>213</fpage>&#x02013;<lpage>229</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-58452-8_13</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>S.-E.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>Y.-C.</given-names></name> <name><surname>Lin</surname> <given-names>E.-T.</given-names></name> <name><surname>Hsiao</surname> <given-names>P.-Y.</given-names></name> <name><surname>Fu</surname> <given-names>L.-C.</given-names></name></person-group> (<year>2023</year>). <article-title>Se-psnet: Silhouette-based enhancement feature for panoptic segmentation network</article-title>. <source>J. Vis. Commun. Image Represent</source>. <volume>90</volume>:<fpage>103736</fpage>. <pub-id pub-id-type="doi">10.1016/j.jvcir.2022.103736</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.-C.</given-names></name> <name><surname>Papandreou</surname> <given-names>G.</given-names></name> <name><surname>Kokkinos</surname> <given-names>I.</given-names></name> <name><surname>Murphy</surname> <given-names>K.</given-names></name> <name><surname>Yuille</surname> <given-names>A. L.</given-names></name></person-group> (<year>2017</year>). <article-title>Deeplab: semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>40</volume>, <fpage>834</fpage>&#x02013;<lpage>848</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2017.2699184</pub-id><pub-id pub-id-type="pmid">28463186</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Q.</given-names></name> <name><surname>Cheng</surname> <given-names>A.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Spatialflow: bridging all tasks for panoptic segmentation</article-title>. <source>IEEE Trans. Circ. Syst. Video Technol</source>. <volume>31</volume>, <fpage>2288</fpage>&#x02013;<lpage>2300</lpage>. <pub-id pub-id-type="doi">10.1109/TCSVT.2020.3020257</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Bourahla</surname> <given-names>O.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Banet: bidirectional aggregation network with occlusion handling for panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 3793&#x02013;3802. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.00385</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>B.</given-names></name> <name><surname>Collins</surname> <given-names>M. D.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Huang</surname> <given-names>T. S.</given-names></name> <name><surname>Adam</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Panoptic-deeplab: A simple, strong, and fast baseline for bottom-up panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 12475&#x02013;12485. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01249</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Condori</surname> <given-names>R. H.</given-names></name> <name><surname>Bruno</surname> <given-names>O. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Analysis of activation maps through global pooling measurements for texture classification</article-title>. <source>Inf. Sci</source>. <volume>555</volume>, <fpage>260</fpage>&#x02013;<lpage>279</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2020.09.058</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cordts</surname> <given-names>M.</given-names></name> <name><surname>Omran</surname> <given-names>M.</given-names></name> <name><surname>Ramos</surname> <given-names>S.</given-names></name> <name><surname>Rehfeld</surname> <given-names>T.</given-names></name> <name><surname>Enzweiler</surname> <given-names>M.</given-names></name> <name><surname>Benenson</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;The cityscapes dataset for semantic urban scene understanding,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, 3213&#x02013;3223. <pub-id pub-id-type="doi">10.1109/CVPR.2016.350</pub-id><pub-id pub-id-type="pmid">32191886</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Geus</surname> <given-names>D.</given-names></name> <name><surname>Meletis</surname> <given-names>P.</given-names></name> <name><surname>Dubbelman</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>Panoptic segmentation with a joint semantic and instance segmentation network</article-title>. <source>arXiv preprint arXiv:1809.02110</source>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>N.</given-names></name> <name><surname>Shan</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Yu</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Ssap: single-shot instance segmentation with affinity pyramid,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>, 642&#x02013;651. <pub-id pub-id-type="doi">10.1109/ICCV.2019.00073</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>N.</given-names></name> <name><surname>Shan</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>Learning category-and instance-aware pixel embedding for fast panoptic segmentation</article-title>. <source>IEEE Trans. Image Proc</source>. <volume>30</volume>, <fpage>6013</fpage>&#x02013;<lpage>6023</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2021.3090522</pub-id><pub-id pub-id-type="pmid">34181542</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Gkioxari</surname> <given-names>G.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Mask R-CNN,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source>, 2961&#x02013;2969. <pub-id pub-id-type="doi">10.1109/ICCV.2017.322</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, 770&#x02013;778. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hong</surname> <given-names>W.</given-names></name> <name><surname>Guo</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Chu</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Lpsnet: a lightweight solution for fast panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 16746&#x02013;16754. <pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01647</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Ren</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Ji</surname> <given-names>R.</given-names></name> <name><surname>Cao</surname> <given-names>L.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;You only segment once: towards real-time panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 17819&#x02013;17829. <pub-id pub-id-type="doi">10.1109/CVPR52729.2023.01709</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Jung</surname> <given-names>A. B.</given-names></name> <name><surname>Wada</surname> <given-names>K.</given-names></name> <name><surname>Crall</surname> <given-names>J.</given-names></name> <name><surname>Tanaka</surname> <given-names>S.</given-names></name> <name><surname>Graving</surname> <given-names>J.</given-names></name> <name><surname>Reinders</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2020</year>). <source>Imgaug</source>. Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/aleju/imgaug">https://github.com/aleju/imgaug</ext-link> (accessed February 01, 2020).</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kirillov</surname> <given-names>A.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name></person-group> (<year>2019a</year>). <article-title>&#x0201C;Panoptic feature pyramid networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 6399&#x02013;6408. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00656</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kirillov</surname> <given-names>A.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Rother</surname> <given-names>C.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name></person-group> (<year>2019b</year>). <article-title>&#x0201C;Panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 9404&#x02013;9413. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00963</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lazarow</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Shi</surname> <given-names>K.</given-names></name> <name><surname>Tu</surname> <given-names>Z.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Learning instance occlusion for panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 10720&#x02013;10729. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01073</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Raventos</surname> <given-names>A.</given-names></name> <name><surname>Bhargava</surname> <given-names>A.</given-names></name> <name><surname>Tagawa</surname> <given-names>T.</given-names></name> <name><surname>Gaidon</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Learning to fuse things and stuff</article-title>. <source>arXiv preprint arXiv:1812.01192</source>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Cheng</surname> <given-names>G.</given-names></name> <name><surname>Shi</surname> <given-names>J.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Improving semantic segmentation via decoupled body and edge supervision,&#x0201D;</article-title> in <source>Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23-28, 2020, Proceedings, Part XVII 16</source> (<publisher-loc>Springer</publisher-loc>), <fpage>435</fpage>&#x02013;<lpage>452</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-58520-4_26</pub-id><pub-id pub-id-type="pmid">37372235</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Xie</surname> <given-names>L.</given-names></name> <name><surname>Huang</surname> <given-names>G.</given-names></name> <name><surname>Du</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Attention-guided unified network for panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, 7026&#x02013;7035. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00719</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>IDNet: information decomposition network for fast panoptic segmentation</article-title>. <source>IEEE Trans. Image Proc</source>. <volume>33</volume>, <fpage>1487</fpage>&#x02013;<lpage>1496</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2023.3234499</pub-id><pub-id pub-id-type="pmid">37037237</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Maire</surname> <given-names>M.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name> <name><surname>Hays</surname> <given-names>J.</given-names></name> <name><surname>Perona</surname> <given-names>P.</given-names></name> <name><surname>Ramanan</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>&#x0201C;Microsoft coco: common objects in context,&#x0201D;</article-title> in <source>Computer Vision-ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13</source> (<publisher-loc>Springer</publisher-loc>), <fpage>740</fpage>&#x02013;<lpage>755</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-10602-1_48</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Peng</surname> <given-names>C.</given-names></name> <name><surname>Yu</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Yu</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;An end-to-end network for panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 6172&#x02013;6181. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00633</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Stathaki</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>Faster R-cnn for robust pedestrian detection using semantic segmentation network</article-title>. <source>Front. Neurorobot</source>. <volume>12</volume>:<fpage>64</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2018.00064</pub-id><pub-id pub-id-type="pmid">30344486</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Min</surname> <given-names>X.</given-names></name> <name><surname>Duan</surname> <given-names>H.</given-names></name> <name><surname>Sun</surname> <given-names>W.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Zhai</surname> <given-names>G.</given-names></name></person-group> (<year>2024</year>). <article-title>Perceptual video quality assessment: a survey</article-title>. <source>arXiv preprint arXiv:2402.03413</source>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Min</surname> <given-names>X.</given-names></name> <name><surname>Zhai</surname> <given-names>G.</given-names></name> <name><surname>Gu</surname> <given-names>K.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name></person-group> (<year>2016</year>). <article-title>Fixation prediction through multimodal analysis</article-title>. <source>ACM Trans. Multim. Comput. Commun. Applic</source>. <volume>13</volume>, <fpage>1</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1145/2996463</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Min</surname> <given-names>X.</given-names></name> <name><surname>Zhai</surname> <given-names>G.</given-names></name> <name><surname>Zhou</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>X.-P.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Guan</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>A multimodal saliency model for videos with high audio-visual correspondence</article-title>. <source>IEEE Trans. Image Proc</source>. <volume>29</volume>, <fpage>3805</fpage>&#x02013;<lpage>3819</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2020.2966082</pub-id><pub-id pub-id-type="pmid">31976898</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mohan</surname> <given-names>R.</given-names></name> <name><surname>Valada</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Efficientps: efficient panoptic segmentation</article-title>. <source>Int. J. Comput. Vis</source>. <volume>129</volume>, <fpage>1551</fpage>&#x02013;<lpage>1579</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-021-01445-z</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sofiiuk</surname> <given-names>K.</given-names></name> <name><surname>Barinova</surname> <given-names>O.</given-names></name> <name><surname>Konushin</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Adaptis: adaptive instance selection network,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, 7355&#x02013;7363. <pub-id pub-id-type="doi">10.1109/ICCV.2019.00745</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>B.</given-names></name> <name><surname>Kuen</surname> <given-names>J.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Mordohai</surname> <given-names>P.</given-names></name> <name><surname>Chen</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;PRN: panoptic refinement network,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision</source>, 3963&#x02013;3973. <pub-id pub-id-type="doi">10.1109/WACV56688.2023.00395</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Takikawa</surname> <given-names>T.</given-names></name> <name><surname>Acuna</surname> <given-names>D.</given-names></name> <name><surname>Jampani</surname> <given-names>V.</given-names></name> <name><surname>Fidler</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Gated-SCNN: gated shape cnns for semantic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>, 5229&#x02013;5238. <pub-id pub-id-type="doi">10.1109/ICCV.2019.00533</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>M.</given-names></name> <name><surname>Pang</surname> <given-names>R.</given-names></name> <name><surname>Le</surname> <given-names>Q. V.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Efficientdet: scalable and efficient object detection,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 10781&#x02013;10790. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01079</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Adam</surname> <given-names>H.</given-names></name> <name><surname>Yuille</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>L.-C.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Max-deeplab: end-to-end panoptic segmentation with mask transformers,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 5463&#x02013;5474. <pub-id pub-id-type="doi">10.1109/CVPR46437.2021.00542</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Green</surname> <given-names>B.</given-names></name> <name><surname>Adam</surname> <given-names>H.</given-names></name> <name><surname>Yuille</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>L.-C.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Axial-deeplab: stand-alone axial-attention for panoptic segmentation,&#x0201D;</article-title> in <source>European Conference on Computer Vision</source> (<publisher-loc>Springer</publisher-loc>), <fpage>108</fpage>&#x02013;<lpage>126</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-58548-8_7</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>R.</given-names></name> <name><surname>Kong</surname> <given-names>T.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Shen</surname> <given-names>C.</given-names></name></person-group> (<year>2020</year>). <article-title>Solov2: Dynamic and fast instance segmentation</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>33</volume>, <fpage>17721</fpage>&#x02013;<lpage>17732</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>Tu</surname> <given-names>Z.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Holistically-nested edge detection,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source>, 1395&#x02013;1403. <pub-id pub-id-type="doi">10.1109/ICCV.2015.164</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>Y.</given-names></name> <name><surname>Liao</surname> <given-names>R.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>R.</given-names></name> <name><surname>Bai</surname> <given-names>M.</given-names></name> <name><surname>Yumer</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Upsnet: a unified panoptic segmentation network,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 8818&#x02013;8826. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00902</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Zhu</surname> <given-names>D.</given-names></name> <name><surname>Zhang</surname> <given-names>G.</given-names></name> <name><surname>Shi</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Contour-aware panoptic segmentation network,&#x0201D;</article-title> in <source>Pattern Recognition and Computer Vision: 4th Chinese Conference, PRCV 2021, Beijing, China, October 29-November 1, 2021, Proceedings, Part II</source>, 79&#x02013;90. <pub-id pub-id-type="doi">10.1007/978-3-030-88007-1_7</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>L.</given-names></name> <name><surname>Lei</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Ye</surname> <given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>Dual-flow network with attention for autonomous driving</article-title>. <source>Front. Neurorobot</source>. <volume>16</volume>:<fpage>978225</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2022.978225</pub-id><pub-id pub-id-type="pmid">36699946</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>T.-J.</given-names></name> <name><surname>Collins</surname> <given-names>M. D.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Hwang</surname> <given-names>J.-J.</given-names></name> <name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Deeperlab: single-shot image parser</article-title>. <source>arXiv preprint arXiv:1902.05093</source>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ye</surname> <given-names>X.</given-names></name> <name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Lei</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Based on cross-scale fusion attention mechanism network for semantic segmentation for street scenes</article-title>. <source>Front. Neurorobot</source>. <volume>17</volume>:<fpage>1204418</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2023.1204418</pub-id><pub-id pub-id-type="pmid">37719330</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Kim</surname> <given-names>D.</given-names></name> <name><surname>Qiao</surname> <given-names>S.</given-names></name> <name><surname>Collins</surname> <given-names>M.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022a</year>). <article-title>&#x0201C;CMT-deeplab: clustering mask transformers for panoptic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 2560&#x02013;2570. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.00259</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Qiao</surname> <given-names>S.</given-names></name> <name><surname>Collins</surname> <given-names>M.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Adam</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2022b</year>). <article-title>&#x0201C;K-means mask transformer,&#x0201D;</article-title> in <source>European Conference on Computer Vision</source> (<publisher-loc>Springer</publisher-loc>), <fpage>288</fpage>&#x02013;<lpage>307</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-19818-2_17</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhai</surname> <given-names>G.</given-names></name> <name><surname>Min</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>Perceptual image quality assessment: a survey</article-title>. <source>Sci. China Inform. Sci</source>. <volume>63</volume>:<fpage>1</fpage>&#x02013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1007/s11432-019-2757-1</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>C.</given-names></name> <name><surname>Xu</surname> <given-names>F.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Xu</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>A lightweight multi-dimension dynamic convolutional network for real-time semantic segmentation</article-title>. <source>Front. Neurorobot</source>. <volume>16</volume>:<fpage>1075520</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2022.1075520</pub-id><pub-id pub-id-type="pmid">36590086</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>C.</given-names></name> <name><surname>Xu</surname> <given-names>F.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Xu</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>Rethinking 1D convolution for lightweight semantic segmentation</article-title>. <source>Front. Neurorobot</source>. <volume>17</volume>:<fpage>1119231</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2023.1119231</pub-id><pub-id pub-id-type="pmid">36845064</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Shi</surname> <given-names>J.</given-names></name> <name><surname>Qi</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Jia</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Pyramid scene parsing network,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, 2881&#x02013;2890. <pub-id pub-id-type="doi">10.1109/CVPR.2017.660</pub-id></citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhen</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Shen</surname> <given-names>T.</given-names></name> <name><surname>Shang</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Joint semantic segmentation and boundary detection using iterative pyramid contexts,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 13666&#x02013;13675. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01368</pub-id></citation>
</ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>H.</given-names></name> <name><surname>Friedman</surname> <given-names>H. S.</given-names></name> <name><surname>Von Der Heydt</surname> <given-names>R.</given-names></name></person-group> (<year>2000</year>). <article-title>Coding of border ownership in monkey visual cortex</article-title>. <source>J. Neurosci</source>. <volume>20</volume>, <fpage>6594</fpage>&#x02013;<lpage>6611</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.20-17-06594.2000</pub-id><pub-id pub-id-type="pmid">10964965</pub-id></citation></ref>
</ref-list>
</back>
</article>