<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1647736</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DSC-DeepLabv3+: a lightweight semantic segmentation model for weed identification in maize fields</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Fu</surname>
<given-names>Haitao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Xiaoyao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3048047/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhu</surname>
<given-names>Li</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pan</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2604928/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Tuo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Wen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Feng</surname>
<given-names>Yuxuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Information Technology, Jilin Agricultural University</institution>, <addr-line>Changchun</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>BaichengAgricultural and Rural Information Center</institution>, <addr-line>Baicheng</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1465011/overview">Zhiwei Ji</ext-link>, Nanjing Agricultural University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2174605/overview">Jingjing Yang</ext-link>, Hebei North University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3118435/overview">Francisco Garibaldi-Marquez</ext-link>, Forestales y Pecuarias, Mexico</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yuxuan Feng, <email xlink:href="mailto:fengyuxuan@jlau.edu.cn">fengyuxuan@jlau.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1647736</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Fu, Li, Zhu, Pan, Wu, Li and Feng.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Fu, Li, Zhu, Pan, Wu, Li and Feng</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Weeds compete with crops for water, nutrients, and light, negatively impacting maize yield and quality. To enhance weed identification accuracy and meet the requirements of precision agriculture, we propose a lightweight semantic segmentation model named DSC-DeepLabv3+.</p>
</sec>
<sec>
<title>Methods</title>
<p>MobileNetV2 is adopted as the backbone, and standard convolutions in atrous spatial pyramid pooling (ASPP) and decoder modules are replaced with depthwise separable dilated convolutions (DSDConv), significantly reducing model complexity and improving segmentation efficiency. To capture rich contextual information, strip pooling is incorporated into the ASPP module, forming the strip pooling&#x2013;atrous spatial pyramid pooling (S-ASPP) structure. In addition, a convolutional block attention module (CBAM) is introduced to refine feature representations, and multi-scale features are further fused using the CBAM&#x2013;Cascade Feature Fusion (C-CFF) module to improve semantic understanding.</p>
</sec>
<sec>
<title>Results</title>
<p>Experimental results show that the proposed model reduces the number of parameters from 54.714M to 2.89M and decreases the computational cost from 167.139 GFLOPs to 15.326 GFLOPs, while achieving an inference speed of 42.89 FPS and a mean Intersection over Union (mIoU) of 85.57%.</p>
</sec>
<sec>
<title>Discussion</title>
<p>These results demonstrate that DSC-DeepLabv3+ strikes an effective balance between accuracy and efficiency, outperforming several classical lightweight models, making it a promising solution for accurate and efficient weed segmentation in agricultural applications.</p>
</sec>
</abstract>
<kwd-group>
<kwd>weed recognition</kwd>
<kwd>lightweight semantic segmentation</kwd>
<kwd>DeepLabV3+</kwd>
<kwd>feature fusion</kwd>
<kwd>attention mechanisms</kwd>
</kwd-group>
<counts>
<fig-count count="12"/>
<table-count count="5"/>
<equation-count count="11"/>
<ref-count count="43"/>
<page-count count="16"/>
<word-count count="7119"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Weeds negatively impact crop yield (<xref ref-type="bibr" rid="B17">Moreau et&#xa0;al., 2022</xref>) and quality (<xref ref-type="bibr" rid="B11">Hamuda et&#xa0;al., 2016</xref>), making effective weed control essential in crop management. Although conventional herbicides are widely used in maize fields due to their ease of application and effectiveness (<xref ref-type="bibr" rid="B42">Zou et&#xa0;al., 2021</xref>), they may they may harm soil health, negatively affect maize, and threaten human health (<xref ref-type="bibr" rid="B18">Muola et&#xa0;al., 2021</xref>). In recent years, weed recognition systems based on machine vision have made progress. However, traditional machine learning methods such as ANN (<xref ref-type="bibr" rid="B25">Shah et&#xa0;al., 2021</xref>), naive Bayes, decision tree, K-means (<xref ref-type="bibr" rid="B1">Agarwal et&#xa0;al., 2021</xref>), and support vector machine (SVM) (<xref ref-type="bibr" rid="B39">Zhang et&#xa0;al., 2022a</xref>) are highly sensitive to environmental variation, which limits their robustness and applicability in precision agriculture (<xref ref-type="bibr" rid="B7">Espejo-Garcia et&#xa0;al., 2020</xref>). With the rapid development of deep learning, convolutional neural networks have demonstrated strong feature learning capabilities (<xref ref-type="bibr" rid="B10">Fuentes-Pacheco et&#xa0;al., 2019</xref>). In particular, semantic segmentation has emerged as a mainstream technique for weed identification (<xref ref-type="bibr" rid="B28">Subeesh et&#xa0;al., 2022</xref>). However, popular models such as FCN (<xref ref-type="bibr" rid="B26">Shelhamer et&#xa0;al., 2017</xref>), Unet (<xref ref-type="bibr" rid="B22">Ronneberger et&#xa0;al., 2015</xref>), DeepLab series (<xref ref-type="bibr" rid="B3">Chen et&#xa0;al., 2016</xref>, <xref ref-type="bibr" rid="B4">2018b</xref>; <xref ref-type="bibr" rid="B24">Sandler et&#xa0;al., 2018</xref>), and PSPNet (<xref ref-type="bibr" rid="B41">Zhao et&#xa0;al., 2017</xref>) are characterized by large parameter sizes, high computational costs, and low inference speed, which limit their practical application and segmentation performance in real-world agricultural settings. To address these limitations, researchers have proposed various lightweight semantic segmentation models designed for real-time applications.</p>
<p>In lightweight semantic segmentation, simplifying the model architecture is a primary objective. Enet (<xref ref-type="bibr" rid="B20">Paszke et&#xa0;al., 2016</xref>) introduced an asymmetric encoder&#x2013;decoder structure, significantly reducing parameter and memory costs and laying the foundation for real-time semantic segmentation. ERFNet (<xref ref-type="bibr" rid="B21">Romera et&#xa0;al., 2018</xref>), EACNet (<xref ref-type="bibr" rid="B15">Li et&#xa0;al., 2021</xref>), and LMFFNet (<xref ref-type="bibr" rid="B27">Shi et&#xa0;al., 2023</xref>) built upon this structure, further reducing parameters through factorized convolutions. The success of SegNet (<xref ref-type="bibr" rid="B2">Badrinarayanan et&#xa0;al., 2017</xref>) demonstrated the effectiveness of encoder&#x2013;decoder architectures with skip connections in resource-constrained environments. DABNet (<xref ref-type="bibr" rid="B16">Li et&#xa0;al., 2019</xref>) and CGNet (<xref ref-type="bibr" rid="B32">Wu et&#xa0;al., 2021</xref>) utilized dilated convolutions to capture both local and contextual features, thereby improving segmentation accuracy. LEDNet (<xref ref-type="bibr" rid="B30">Wang et&#xa0;al., 2019</xref>) and LAANet (<xref ref-type="bibr" rid="B38">Zhang et&#xa0;al., 2022b</xref>) incorporated attention mechanisms to enhance contextual feature representation and boost performance. The ICNet (<xref ref-type="bibr" rid="B40">Zhao et&#xa0;al., 2018</xref>) adopted a cascaded multi-resolution structure to balance segmentation accuracy and real-time efficiency. BiSeNet (<xref ref-type="bibr" rid="B37">Yu et&#xa0;al., 2018</xref>) introduced a dual-branch architecture to separately extract spatial and semantic features. BiSeNetV2 (<xref ref-type="bibr" rid="B36">Yu et&#xa0;al., 2021</xref>) and STDC (<xref ref-type="bibr" rid="B8">Fan et&#xa0;al., 2021</xref>) further optimized efficiency through feature-sharing designs. Although PIDNet (<xref ref-type="bibr" rid="B34">Xu et&#xa0;al., 2023</xref>) and DDRNet (<xref ref-type="bibr" rid="B19">Pan et&#xa0;al., 2023</xref>) improved performance by employing multi-path strategies, they introduced additional computational burdens. Recent trends in lightweight segmentation focus on innovative architectures, efficient feature fusion modules, and weakly or unsupervised learning methods to enhance adaptability and performance.</p>
<p>In recent years, lightweight semantic segmentation algorithms have shown great potential in practical applications. In the field of crop&#x2013;weed and remote sensing image segmentation, Zuo et&#xa0;al (<xref ref-type="bibr" rid="B43">Zuo and Li, 2024</xref>). proposed a lightweight U-Net variant, which enhances cornfield weed segmentation efficiency by incorporating an inverted residual structure, pyramid pooling, and a squeeze-and-excitation mechanism. Sun et&#xa0;al (<xref ref-type="bibr" rid="B29">Sun et&#xa0;al., 2025</xref>). proposed ASLMSHNet, which optimizes feature fusion and resource allocation for remote sensing image segmentation through progressive dilated convolutions, adaptive sparse cross-attention, and multi-scale feature alignment. Janneh et&#xa0;al (<xref ref-type="bibr" rid="B14">Janneh et&#xa0;al., 2023</xref>). introduced a multilevel feature reweighting framework that comprises a lightweight backbone, a reweighting fusion module, and a convolutional weighted decoder. This approach reduces feature dimensionality, suppresses background interference, and improves both contextual understanding and segmentation efficiency. In the domain of plant disease and infrastructure defect detection, Feng et&#xa0;al (<xref ref-type="bibr" rid="B9">Feng et&#xa0;al., 2022</xref>). introduced DFFANet, which integrates deep feature fusion and attention mechanisms through modules such as DCABlock, FFM, and an efficient attention module. This design ensures accurate segmentation of rice blast spots while maintaining low model complexity. Yu et&#xa0;al (<xref ref-type="bibr" rid="B35">Yu et&#xa0;al., 2025</xref>). improved the DeepLabv3+ model for bridge deck disease detection by incorporating MobileNetV3, a CSF-ASPP module, and a focal loss function. These modifications significantly reduced parameter count and computational complexity while enhancing the recognition accuracy of small-scale disease regions. While these models demonstrate strong performance in specific scenarios, lightweight networks often suffer from accuracy degradation when reducing parameters or accelerating inference. Moreover, most existing methods are tailored to specific domains, limiting their generalization and adaptability across diverse field conditions. In addition to architectural improvements, recent studies have explored optimization strategies at the feature level to further enhance model performance. For instance, Xie et&#xa0;al (<xref ref-type="bibr" rid="B33">Xie et&#xa0;al., 2023</xref>). proposed a feature selection strategy based on the Salp Swarm Algorithm for plant disease detection. Such approaches highlight the potential of bio-inspired algorithms in reducing model complexity while maintaining performance.</p>
<p>We chose the DeepLabv3+ (<xref ref-type="bibr" rid="B4">Chen et&#xa0;al., 2018</xref>) model as the basic framework due to its excellent performance in semantic segmentation. However, the model has several limitations: its feature extraction network is overly complex and contains a large number of parameters. Additionally, the use of standard convolutions in the ASPP module further increases the parameter count, thereby increasing model complexity, hardware requirements, and reducing training efficiency. Moreover, the encoder stage progressively reduces the spatial resolution of the input, leading to information loss and insufficient restoration of fine details during decoding. As a result, boundary localization remains suboptimal. Although the ASPP module enhances boundary extraction, it fails to adequately model local feature relationships, leading to fragmented segmentation and reduced accuracy, particularly at object edges. To address these limitations and achieve improved accuracy, lightweight design, and faster inference, we propose an enhanced version of DeepLabv3+. The primary contributions of this study are summarized as follows:</p>
<list list-type="order">
<list-item>
<p>The original network has been substituted using MobileNetV2, while standard convolutions in the encoder-decoder segments were substituted with depthwise separable dilated convolutions, thereby reducing computational load and training time.</p>
</list-item>
<list-item>
<p>To accurately capture distant dependencies and acquire dense contextual information, the strip pooling is integrated within the ASPP module, and CBAM is applied after ASPP to enhance feature maps&#x2019; capacity to extract detailed information.</p>
</list-item>
<list-item>
<p>To fully utilize features from the two intermediate layers and improve segmentation accuracy, in the decoder part, we propose the C-CFF module for feature fusion.</p>
</list-item>
</list>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Construction of the maize weed dataset</title>
<sec id="s2_1_1">
<label>2.1.1</label>
<title>Data acquisition</title>
<p>Maize weed images were collected at Jilin Agricultural University (Changchun, Jilin Province, China) between 10:30 and 14:30 on June 10 and June 20, 2024. Data were captured using a Xiaomi 14 smartphone equipped with a 50 MP rear camera (ISO 50, shutter speed: 1/200 s), mounted vertically at a height of 50 cm above the ground. Videos with a resolution of 1280 &#xd7; 720 pixels were recorded and subsequently converted into individual JPG images of the same resolution.</p>
</sec>
<sec id="s2_1_2">
<label>2.1.2</label>
<title>Data preprocessing</title>
<p>After removing unusable samples, a final set of 481 valid images containing both maize seedlings and weeds was retained. These images were subsequently annotated using LabelMe (<xref ref-type="bibr" rid="B23">Russell et&#xa0;al., 2008</xref>), with each pixel classified into three categories: &#x201c;corn&#x201d; for maize seedlings, &#x201c;weed&#x201d; for weeds, and background for all other regions. Data augmentation was employed to enhance the model&#x2019;s robustness, with the specific techniques illustrated in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. Following augmentation, the dataset was expanded to 2,886 images, ensuring sufficient diversity for training and reliable performance evaluation. The dataset was randomly split into training and validation subsets at a 9:1 ratio, and all images were subsequently converted to Pascal VOC format for use in this study.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Examples of data augmentation. <bold>(A)</bold> Original image. <bold>(B)</bold> Flip. <bold>(C)</bold> Rotation. <bold>(D)</bold> Gaussian Blur. <bold>(E)</bold> Random Crop. <bold>(F)</bold> HSV.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g001.tif">
<alt-text content-type="machine-generated">Six-panel image showing plants with variations in leaf health and color against a soil background. Panels A, B, D, and E display green, healthy-looking leaves. Panel C has slightly tilted leaves, while Panel F shows yellowing leaves, indicating possible distress. Each panel is labeled alphabetically.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>DeepLabv3+ model structure</title>
<p>As shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, DeepLabv3+ is a semantic segmentation model developed by Google that employs an encoder&#x2013;decoder architecture to integrate multi-scale contextual features and fine spatial details. It uses the Xception network (<xref ref-type="bibr" rid="B6">Chollet, 2017</xref>) as its feature extraction backbone and integrates an ASPP module, which combines a 1&#xd7;1 convolution with three parallel 3&#xd7;3 dilated convolutions at different dilation rates to capture multi-scale contextual information. The decoder fuses high-level semantic features with low-level edge information, enabling accurate pixel-wise segmentation and enhancing boundary delineation in small objects and complex scenes.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>The original DeepLabv3+ model structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g002.tif">
<alt-text content-type="machine-generated">Diagram of an image processing network. It shows an encoder-decoder structure. The encoder includes a Backbone Network and Atrous Spatial Pyramid Pooling (ASPP) module with various convolution layers and image pooling. The decoder includes convolution and concatenation layers, followed by upsampling, leading to the output image.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Improved DeepLabv3+ semantic segmentation model</title>
<p>Although DeepLabv3+ exhibits strong performance in semantic segmentation, certain limitations remain that hinder its broader applicability. The encoder reduces the spatial resolution of feature maps and fails to fully exploit high-resolution information, leading to discontinuities in predictions and a loss of fine-grained details. While the ASPP module enhances boundary feature extraction, it inadequately captures local structural information, particularly for radially distributed objects such as weeds and maize leaves, resulting in fragmented segmentation and semantic gaps. Moreover, the Xception backbone introduces a substantial computational burden, increasing hardware demands and reducing training efficiency. To address these limitations, this study proposes a lightweight improved DSC-DeepLabv3+ model that optimizes the encoder&#x2013;decoder architecture and enhances both computational efficiency and segmentation accuracy. An overview of the proposed model is illustrated in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. In the encoder, the original Xception backbone is replaced with MobileNetV2 to achieve efficient feature extraction. The ASPP module is enhanced with depthwise separable convolutions to reduce parameters and improve training efficiency. To further enrich global and local contextual information, a strip pooling module is integrated, resulting in a modified S-ASPP structure with six parallel branches. Additionally, the CBAM is incorporated to enhance segmentation accuracy while maintaining low computational complexity. On the decoder side, the C-CFF module is employed to fuse 1/8 and 1/16 scale feature maps extracted from the backbone. CBAM is introduced again to suppress redundant noise and alleviate edge blurring. Shallow features are fused and upsampled via bilinear interpolation to restore spatial details. The fused deep and multi-scale features are subsequently processed through a 3&#xd7;3 convolution and fourfold upsampling to restore the original image resolution.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Improved DSC-DeepLabv3+ model structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g003.tif">
<alt-text content-type="machine-generated">Diagram illustrating a deep learning model for plant segmentation. On the left, an input image shows a plant. The encoder with MobileNetV2 processes the image through multiple layers, represented by colored blocks. S-ASPP handles spatial information with DSDConv layers and pooling methods. The C-CFF component interacts between layers. The decoder reconstructs the image with DSDConv and unsampling steps, ending with a segmented output image on the right showing the plant in red against a dark background.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Backbone Network</title>
<p>The original DeepLabv3+ employed the Xception network as its feature extraction backbone. Although effective on large-scale datasets, its performance degrades in practical applications characterized by limited annotated data. Its large parameter count and complex architecture result in slow inference and extended training time, hindering deployment on mobile and embedded platforms. To address these limitations, this study proposes a lightweight model by adopting MobileNetV2 (<xref ref-type="bibr" rid="B24">Sandler et&#xa0;al., 2018</xref>) as the backbone. MobileNetV2 was introduced by Google in 2018, featuring a streamlined architecture optimized for mobile and embedded devices. It was designed to offer an efficient solution for various visual recognition tasks, including object classification and image segmentation. The architectural details of the backbone network are presented in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The core innovation of MobileNetV2 lies in replacing standard convolutions with depthwise separable convolutions. This approach significantly reduces computational cost and model size by decomposing standard convolutions into two operations: depthwise convolution and pointwise convolution. In this structure, the depthwise operation applies a 3&#xd7;3 convolution to each input channel independently, while the pointwise operation performs a 1&#xd7;1 convolution to aggregate information across channels and produce the final feature map. This design improves training efficiency and reduces the computational cost of 3&#xd7;3 convolutions by up to 90%, while maintaining high segmentation accuracy. However, despite these improvements, feature extraction remains suboptimal due to kernel sparsity, with a large proportion of convolutional parameters contributing minimally. To address these issues, MobileNetV2 introduces the linear bottleneck and inverted residual structure, as illustrated in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. The linear bottleneck reduces parameter count and computational overhead by eliminating nonlinear activations that may distort low-dimensional feature representations. The inverted residual structure, incorporating designs with two different strides, facilitates network deepening, prevents the vanishing gradient problem, and decreases the number of parameters. Collectively, these innovations allow MobileNetV2 to achieve high performance while maintaining low computational complexity.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>The structure of the backbone.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Input</th>
<th valign="middle" align="center">Operator</th>
<th valign="middle" align="center">t</th>
<th valign="middle" align="center">c</th>
<th valign="middle" align="center">n</th>
<th valign="middle" align="center">s&#x200b;</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">512&#xd7;512&#xd7;3</td>
<td valign="middle" align="center">conv2d</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">32</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">2</td>
</tr>
<tr>
<td valign="middle" align="center">256&#xd7;256&#xd7;32</td>
<td valign="middle" align="center">Bottleneck</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">16</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">1</td>
</tr>
<tr>
<td valign="middle" align="center">256&#xd7;256&#xd7;16</td>
<td valign="middle" align="center">Bottleneck</td>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">24</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">2</td>
</tr>
<tr>
<td valign="middle" align="center">128&#xd7;128&#xd7;24</td>
<td valign="middle" align="center">Bottleneck</td>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">32</td>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">2</td>
</tr>
<tr>
<td valign="middle" align="center">64&#xd7;64&#xd7;32</td>
<td valign="middle" align="center">Bottleneck</td>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">64</td>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">2</td>
</tr>
<tr>
<td valign="middle" align="center">32&#xd7;32&#xd7;64</td>
<td valign="middle" align="center">Bottleneck</td>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">96</td>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">1</td>
</tr>
<tr>
<td valign="middle" align="center">32&#xd7;32&#xd7;96</td>
<td valign="middle" align="center">Bottleneck</td>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">160</td>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">2</td>
</tr>
<tr>
<td valign="middle" align="center">16&#xd7;16&#xd7;160</td>
<td valign="middle" align="center">Bottleneck</td>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">320</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">1</td>
</tr>
<tr>
<td valign="middle" align="center">16&#xd7;16&#xd7;320</td>
<td valign="middle" align="center">conv2d</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">1280</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">1</td>
</tr>
<tr>
<td valign="middle" align="center">16&#xd7;16&#xd7;1280</td>
<td valign="middle" align="center">AvgPool</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">1280</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center"/>
</tr>
<tr>
<td valign="middle" align="center">1&#xd7;1&#xd7;1280</td>
<td valign="middle" align="center">Classifier</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">k</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>t, is the expansion factor in Bottleneck; c, is the number of output channels; n, is the number of repetitions of the operation; s, is the step size.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>The inverted residual and linear bottleneck structure. <bold>(A)</bold> Inverted residual block with a stride of 1. <bold>(B)</bold> Inverted residual block with a stride of 2.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g004.tif">
<alt-text content-type="machine-generated">Diagram comparing two neural network blocks labeled A and B. Block A has input passing through Conv2d 1&#xd7;1 with ReLU6, then DWConv 3&#xd7;3 with ReLU6, and Conv2d 1&#xd7;1 Linear, followed by adding the input with stride 1. Block B has input processing through Conv2d 1&#xd7;1 with ReLU6, DWConv 3&#xd7;3 with ReLU6, and Conv2d 1&#xd7;1 Linear, with stride 2.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>S-ASPP</title>
<p>This study enhances training efficiency by introducing depthwise separable dilated convolutions, which integrate the benefits of both dilated and depthwise separable operations, as illustrated in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. In the depthwise stage, each channel of the feature map is convolved independently with a dilated kernel to extract spatial correlations while preserving local structural information. Subsequently, a 1&#xd7;1 pointwise convolution aggregates the features across channels to generate the final output. This method enlarges the receptive field, reduces parameter count and computational complexity, and accelerates inference. Additionally, the global average pooling in the traditional ASPP (<xref ref-type="bibr" rid="B12">He et&#xa0;al., 2015</xref>) employs a fixed-size square window, which presents certain limitations. Such a design struggles to capture directional scale correlations when processing irregularly shaped objects or complex environments. The square pooling window may introduce redundant dependencies and incorporate noise from unrelated regions, leading to the loss of critical fine-grained information. To overcome these limitations, Strip Pooling (<xref ref-type="bibr" rid="B13">Hou et&#xa0;al., 2020</xref>) is integrated into the ASPP module, resulting in a novel six-branch S-ASPP structure. This modification reduces model complexity, enhances inference speed, and enables the capture of diverse multi-scale contextual features. Strip pooling performs directional pooling along horizontal and vertical dimensions to simultaneously capture global and local contextual information while suppressing background noise. It utilizes one-dimensional convolutions along each direction within a residual framework to enhance directional sensitivity. The resulting feature maps are fused and element-wise multiplied with the original features to refine spatial representations. Unlike traditional square pooling, strip pooling independently processes vertical and horizontal spatial dimensions by performing weighted averaging across rows and columns. The structure of the strip pooling module is illustrated in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The structure of depthwise separable dilated convolution.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g005.tif">
<alt-text content-type="machine-generated">Diagram showing a neural network architecture with two sections: DWConv (Depthwise Convolution) and PWConv (Pointwise Convolution). The DWConv section uses blue squares and orange grids. The PWConv section involves yellow 3D cubes transforming into white squares. Arrows indicate data flow through the architecture.</alt-text>
</graphic>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>The structure of strip pooling.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g006.tif">
<alt-text content-type="machine-generated">Diagram of a neural network operation showing an input tensor undergoing strip pooling in horizontal and vertical directions, processed by 1D convolution, fusion, and a one-by-one convolution followed by a sigmoid function. The result is combined with the identity path to produce the output tensor.</alt-text>
</graphic>
</fig>
<p>For the given input image, the two vectors of the input picture are computed using <xref ref-type="disp-formula" rid="eq1">Equations 1</xref>, <xref ref-type="disp-formula" rid="eq2">2</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>W</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>W</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>v</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>H</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>H</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>For an input <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, the total amount of channels is denoted by <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the row and column respectively, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the size, <inline-formula>
<mml:math display="inline" id="im4">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula> is passed via both vertical and horizontal paths for pooling. The horizontal and vertical outputs are <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msup>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>y</mml:mi>
</mml:mstyle>
<mml:mi>v</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Following their combination, the <xref ref-type="disp-formula" rid="eq3">Equation 3</xref> is used to figure out the final output:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>h</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>v</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>A refined feature map is obtained after completing the convolution operations and applying the sigmoid activation function, and then it is merged with the input feature map for the production of <inline-formula>
<mml:math display="inline" id="im6">
<mml:mi>z</mml:mi>
</mml:math>
</inline-formula>. The specific process is shown in <xref ref-type="disp-formula" rid="eq4">Equation 4</xref>:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>Scale</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Improved C-CFF module</title>
<sec id="s2_6_1">
<label>2.6.1</label>
<title>CFF module</title>
<p>The original Cascade Feature Fusion (CFF) module in ICNet (<xref ref-type="bibr" rid="B40">Zhao et&#xa0;al., 2018</xref>) improved semantic segmentation by fusing multi-resolution features to enhance spatial detail preservation. Specifically, it accepts a shallow feature map F<sub>1</sub> and a deep feature map F<sub>2</sub> as inputs. First, F<sub>2</sub> is upsampled by a factor of two using bilinear interpolation, followed by a dilated convolution to achieve spatial alignment with F<sub>1.</sub> A 1&#xd7;1 convolution is then applied to F<sub>1</sub> to adjust its channel dimensions to match those of the processed F<sub>2.</sub> Both feature maps are normalized through batch normalization layers, then fused via element-wise addition. The resulting map is passed through a ReLU activation function to obtain the final fused feature map F<sub>c</sub>. However, the simple element-wise addition disregards the heterogeneity in semantic and spatial information between the features, which limits its effectiveness in tasks requiring fine-grained segmentation, such as leaf edge and small object recognition. Furthermore, this naive fusion strategy may introduce redundant noise across both channel-wise and spatial dimensions. In complex agricultural scenarios that demand precise localization of crop and weed boundaries, the original CFF module lacks the capacity to emphasize salient regions, potentially leading to blurred boundaries and loss of detailed features.</p>
</sec>
<sec id="s2_6_2">
<label>2.6.2</label>
<title>CBAM</title>
<p>In recent years, attention mechanisms have become widely used in computer vision tasks due to their ability to selectively focus on salient regions and efficiently capture informative visual cues. Such mechanisms have been increasingly integrated into convolutional neural networks to enhance performance in large-scale image classification tasks. CBAM (<xref ref-type="bibr" rid="B31">Woo et&#xa0;al., 2018</xref>) is a lightweight attention module that combines both channel and spatial attention to significantly improve model accuracy while introducing minimal computational overhead. It can be seamlessly embedded into various convolutional neural network architectures without requiring extensive modifications. As illustrated in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>, CBAM consists of two sequential submodules. It generates attention maps by analyzing intermediate feature representations to emphasize salient features and suppress irrelevant information, thereby improving the network&#x2019;s ability to extract meaningful patterns from complex visual data.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>
<bold>(A)</bold> Convolutional block attention module. <bold>(B)</bold> Channel attention module. <bold>(C)</bold> Spatial attention module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g007.tif">
<alt-text content-type="machine-generated">Diagram illustrating three modules: (A) Convolutional Block Attention Module integrating inputs from Channel and Spatial Attention Modules; (B) Channel Attention Module using MaxPool and AvgPool, followed by MLP to produce CAM Output Feature; (C) Spatial Attention Module using Channel Refined Features with MaxPool and AvgPool to produce SAM Output Feature.</alt-text>
</graphic>
</fig>
<p>This channel attention part is designed based on treating every channel as a distinct feature detector. Specifically, spatial information within representations of features is first compressed using two global pooling processes, resulting in channel attention representations. Vectors are passed through an MLP, resulting in two attention vectors of size <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. By element-wise summing the outputs and applying a sigmoid activation function, a final channel attention vector <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of size <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is obtained. Its calculation formula is shown in <xref ref-type="disp-formula" rid="eq5">Equation 5</xref>:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mtext>avg</mml:mtext>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im10">
<mml:mi>F</mml:mi>
</mml:math>
</inline-formula> is the given middle feature, c denotes the channel dimension,avg denotes global average pooling, max denotes maximum pooling, <inline-formula>
<mml:math display="inline" id="im11">
<mml:mi>&#x3c3;</mml:mi>
</mml:math>
</inline-formula> denotes Sigmoid, and <inline-formula>
<mml:math display="inline" id="im12">
<mml:mi>&#x3c9;</mml:mi>
</mml:math>
</inline-formula> represents the fully connected operation.</p>
</sec>
<sec id="s2_6_3">
<label>2.6.3</label>
<title>CBAM-cascade feature fusion module</title>
<p>Motivated by the effectiveness of attention mechanisms in enhancing feature selection, we incorporated the CBAM module into the CFF structure. Specifically, CBAM is positioned after the fusion of the feature maps F<sub>1</sub> and F<sub>2</sub>, enabling refined recalibration of the fused features. It adaptively recalibrates the importance of each semantic channel, thereby enhancing the representation of fine-scale targets such as crop seedlings. Moreover, its spatial attention branch emphasizes high-frequency regions such as leaf boundaries, improving the network&#x2019;s capacity to capture fine-grained spatial details. This strategy suppresses background interference while enhancing the boundary sensitivity of the segmentation map. As a result, the network dynamically balances feature responses along both channel and spatial dimensions, facilitating accurate boundary recovery without compromising computational efficiency. The architecture of the enhanced fusion module is illustrated in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>. This improved feature fusion approach preserves shallow feature richness while fully utilizing deep semantic representations, thereby enhancing both the segmentation accuracy and the robustness of the network.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Structure of the improved C-CFF module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g008.tif">
<alt-text content-type="machine-generated">Flowchart of a neural network model showing two input features, F&#x2081; and F&#x2082;. F&#x2081; undergoes a 1x1 convolution followed by batch normalization. F&#x2082; is upsampled by 2, processed with a 3x3 atrous convolution with rate 2, and batch normalized. The results are summed and passed through a Convolutional Block Attention Module (CBAM), labeled as the improved part, followed by a ReLU activation function producing output F&#x209b;.</alt-text>
</graphic>
</fig>
<p>For feature maps <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, where the size of F<sub>1</sub> is twice that of F<sub>2</sub>. We first apply an upsampling rate of 2 on F<sub>2</sub> through bilinear interpolation, followed by dilated convolution to keep same as F<sub>1</sub>. Meanwhile, F<sub>1</sub> conducts a 1x1 convolution operation to achieve the same number of feature channels as F<sub>2</sub>. These two processed features are then normalized using two batch normalization layers. Finally, the two features were added together to obtain F<sub>3</sub> as described by <xref ref-type="disp-formula" rid="eq6">Equation 6</xref>:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3ba;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im15">
<mml:mi>&#x3ba;</mml:mi>
</mml:math>
</inline-formula> denotes a 1 &#xd7; 1 convolution, <inline-formula>
<mml:math display="inline" id="im16">
<mml:mi>&#x3d5;</mml:mi>
</mml:math>
</inline-formula> is a dilation convolution, <inline-formula>
<mml:math display="inline" id="im17">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> denotes upsampling, and <inline-formula>
<mml:math display="inline" id="im18">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula> denotes batch normalisation.</p>
<p>Subsequently processed by the CBAM module, the channel attention weights were first generated by the channel attention module <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The channel attention map <inline-formula>
<mml:math display="inline" id="im20">
<mml:mi>G</mml:mi>
</mml:math>
</inline-formula> is obtained by multiplying this weight by the input features. Then, the spatial attention module was used to generate the spatial attention weights <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and get the spatial attention feature map <inline-formula>
<mml:math display="inline" id="im22">
<mml:mi>&#x3c4;</mml:mi>
</mml:math>
</inline-formula>, which is finally activated by the ReLU and generates the final output <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Specific calculation steps are calculated via <xref ref-type="disp-formula" rid="eq7">Equations 7</xref>&#x2013;<xref ref-type="disp-formula" rid="eq9">9</xref>:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:msubsup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mtext>avg</mml:mtext>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2299;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3ba;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2299;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im24">
<mml:mo>&#x2299;</mml:mo>
</mml:math>
</inline-formula> denotes element-wise multiplication, <inline-formula>
<mml:math display="inline" id="im25">
<mml:mi>G</mml:mi>
</mml:math>
</inline-formula> denotes the feature after channel attention processing, <inline-formula>
<mml:math display="inline" id="im26">
<mml:mi>&#x3c4;</mml:mi>
</mml:math>
</inline-formula> denotes the spatial attention feature map after processing, and <inline-formula>
<mml:math display="inline" id="im27">
<mml:mi>&#x3b4;</mml:mi>
</mml:math>
</inline-formula> denotes the ReLU activation function.</p>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Evaluation metrics</title>
<p>The purpose of this paper is to keep a lightweight model while obtaining outstanding segmentation accuracy. We evaluate the model&#x2019;s performance using widely utilized semantic segmentation metrics such as mIoU. The calculation of the mIoU is defined by <xref ref-type="disp-formula" rid="eq10">Equation 10</xref>:</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mtext>mIoU</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>FP</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where n is the number of classes, TP is true positive, TN is true negative, FP is false positive, and FN is false negative.</p>
<p>Parameters reflect the size of the model. FLOPs (Floating Point Operations) are its computational complexity. FPS measures processing speed, representing the time taken to process a picture.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Model training</title>
<p>The semantic segmentation model was implemented in the PyTorch framework under the following software environment: Python 3.8.19, PyTorch 2.4.0, CUDA 11.8, and Windows 11. The experiments were conducted on a workstation equipped with an AMD 7745HX CPU and an NVIDIA GeForce RTX 4060 GPU.</p>
<p>During training, input images were resized to 512&#xd7;512 pixels. Stochastic Gradient Descent (SGD) optimizer was adopted as the optimizer, with an initial learning rate of 0.007, a minimum learning rate set to 0.01 of the maximum, and a weight decay of 0.0001. The training process lasted for 300 epochs, with the first 100 epochs conducted under a frozen backbone using a batch size of 8, and the remaining 200 epochs under an unfrozen backbone with a batch size of 4. Model validation and checkpoint saving were performed every 20 epochs. In the dataset, maize occupies the majority of the pixel area, resulting in significant foreground-background class imbalance. To mitigate the adverse effects of this imbalance on segmentation performance, the cross-entropy loss function was employed, as defined in <xref ref-type="disp-formula" rid="eq11">Equation 11</xref>:</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mtext>Cross_entropy</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im28">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> is the overall count of samples, <inline-formula>
<mml:math display="inline" id="im29">
<mml:mi>M</mml:mi>
</mml:math>
</inline-formula> symbolizes how many classes, <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the true value of the i-th sample belonging to class c, and <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the model&#x2019;s projected likelihood that the i-th sample falls into class c.</p>
<p>Thanks to the pre-trained backbone adopted through transfer learning, the training loss converged rapidly to a low value. <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> depicts the mIoU and loss curves of the proposed model. After 300 epochs, the mean Intersection over Union (mIoU) reached 85.47%. As shown in the loss curve, the validation and training losses decreased to approximately 0.1 and 0.2, respectively, with minimal fluctuations, indicating stable convergence. Beyond this point, further training yielded marginal improvements in loss, suggesting that the model had reached optimal convergence.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Model training results. <bold>(A)</bold> Training mIoU change curve. <bold>(B)</bold> Training loss curve.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g009.tif">
<alt-text content-type="machine-generated">Graph A illustrates the mIoU curve over 300 epochs, showing an increase from near 0 to over 80 percent, stabilizing thereafter. Graph B presents the loss curve with multiple data lines&#x2014;train loss, validation loss, smooth train loss, and smooth validation loss&#x2014;all rapidly decreasing to below 0.2 within 50 epochs, then leveling off.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Comparison of various models for semantic segmentation</title>
<p>To evaluate the performance of the proposed method in maize field weed recognition, we compared it with both classical and lightweight models, including SegNet, BiSeNet, and ICNet. All models were trained under identical experimental conditions and preprocessing procedures. As shown by the loss curves in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, the proposed model demonstrates a faster convergence rate and achieves lower final loss values within 300 training epochs for both the training and validation sets. Compared to other models, it also shows a steeper initial loss descent and faster overall convergence, effectively reducing the required training time.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Loss comparison plots. <bold>(A)</bold> Training loss plots. <bold>(B)</bold> Validation loss plots.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g010.tif">
<alt-text content-type="machine-generated">Line graphs labeled A and B compare the loss of various neural network models across 300 epochs. Models include DeepLabV3+, UNet, PSPNet, FCN, SegNet, BiSeNet, ICNet, and a custom model labeled &#x201c;Ours&#x201d;. Both graphs show a significant decrease in loss initially, stabilizing over time. The legend differentiates models by color.</alt-text>
</graphic>
</fig>
<p>This study compares the accuracy and computational complexity of various models, as summarized in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. The proposed model reduced FLOPs to 15.326G, which is approximately 150G less than the original DeepLabv3+, while incurring only a 0.71% decrease in mean Intersection over Union (mIoU). Its parameter count was reduced to just 5% of the original, and inference speed increased by 24.49 FPS, making it well-suited for mobile deployment. Compared to SegNet and ICNet, the proposed model achieved mIoU improvements of 7.67% and 5.92%, respectively. It also demonstrated notable gains in inference speed (33.44 and 12.59 FPS), while reducing parameters by 26.55M and 23.61M, and FLOPs by 111G and 12.974G. Compared to BiSeNet, it achieved slightly higher inference speed and a 7.1% increase in mIoU, striking a balance among model compactness, computational efficiency, and segmentation performance.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison of segmentation results of several methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Backbone</th>
<th valign="middle" align="center">mIoU (%)</th>
<th valign="middle" align="center">Parameters (M)</th>
<th valign="middle" align="center">FLOPs (G)</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">FCN</td>
<td valign="middle" align="center">VGG16</td>
<td valign="middle" align="center">75.0</td>
<td valign="middle" align="center">32.75</td>
<td valign="middle" align="center">89.8</td>
<td valign="middle" align="center">14.73</td>
</tr>
<tr>
<td valign="middle" align="center">Unet</td>
<td valign="middle" align="center">VGG16</td>
<td valign="middle" align="center">79.2</td>
<td valign="middle" align="center">43.93</td>
<td valign="middle" align="center">184.4</td>
<td valign="middle" align="center">27.76</td>
</tr>
<tr>
<td valign="middle" align="center">PSPNet</td>
<td valign="middle" align="center">ResNet50</td>
<td valign="middle" align="center">74.7</td>
<td valign="middle" align="center">46.716</td>
<td valign="middle" align="center">118.47</td>
<td valign="middle" align="center">29.4</td>
</tr>
<tr>
<td valign="middle" align="center">DeepLabv3+</td>
<td valign="middle" align="center">Xception</td>
<td valign="middle" align="center">
<bold>86.28</bold>
</td>
<td valign="middle" align="center">54.714</td>
<td valign="middle" align="center">167.139</td>
<td valign="middle" align="center">18.4</td>
</tr>
<tr>
<td valign="middle" align="center">SegNet</td>
<td valign="middle" align="center">VGG16</td>
<td valign="middle" align="center">77.9</td>
<td valign="middle" align="center">29.44</td>
<td valign="middle" align="center">126.34</td>
<td valign="middle" align="center">9.45</td>
</tr>
<tr>
<td valign="middle" align="center">BiSeNet</td>
<td valign="middle" align="center">Xception</td>
<td valign="middle" align="center">78.47</td>
<td valign="middle" align="center">5.8</td>
<td valign="middle" align="center">50.3</td>
<td valign="middle" align="center">41.23</td>
</tr>
<tr>
<td valign="middle" align="center">ICNet</td>
<td valign="middle" align="center">ResNet50</td>
<td valign="middle" align="center">79.65</td>
<td valign="middle" align="center">26.5</td>
<td valign="middle" align="center">28.3</td>
<td valign="middle" align="center">30.3</td>
</tr>
<tr>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">MobileNetV2</td>
<td valign="middle" align="center">85.57</td>
<td valign="middle" align="center">
<bold>2.89</bold>
</td>
<td valign="middle" align="center">
<bold>15.326</bold>
</td>
<td valign="middle" align="center">
<bold>42.89</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Performance comparison of different models. Bold values indicate the best results.</p>
</table-wrap-foot>
</table-wrap>
<p>
<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> presents the segmentation results of various models on the maize weed dataset. The proposed model demonstrates superior performance in boundary delineation and pixel-level classification, particularly in scenarios involving overlapping maize and weeds with incomplete or ambiguous shape features. In contrast, models such as PSPNet, FCN, BiSeNet, and ICNet exhibit misclassification of adjacent pixels, primarily due to insufficient global contextual modeling. For instance, the pyramid pooling module in PSPNet compromises spatial detail, while FCN&#x2019;s limited receptive field overly emphasizes local features, leading to errors in blurred boundary regions. Although BiSeNet and ICNet achieve faster inference through multi-branch architectures, their aggressive downsampling and coarse feature fusion reduce semantic consistency, particularly affecting sensitivity to small objects. U-Net and DeepLabv3+ also suffer from imprecise edge segmentation. In U-Net, skip connections introduce noise from shallow layers, which, when fused with deep features, contribute to edge blurring. Standard upsampling operations further smooth the boundaries, undermining the recovery of fine-grained details. Although the ASPP module in DeepLabv3+ expands the receptive field, its reliance on standard convolutions and global pooling reduces responsiveness to high-frequency edge features. Additionally, the lack of targeted enhancement mechanisms during low-resolution feature fusion in the decoder leads to boundary deviations from the actual object contours. In contrast, the proposed model incorporates strip pooling to capture long-range contextual dependencies and mitigate pixel misclassification in overlapping regions. Furthermore, a feature fusion module enhances the recovery of essential spatial details for precise boundary localization. This architecture effectively balances segmentation accuracy and model efficiency, making it highly suitable for real-time maize weed recognition in complex field environments.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Segmentation results of various models on the maize weed dataset. <bold>(A)</bold> Original image. <bold>(B)</bold> Label image. <bold>(C)</bold> Unet. <bold>(D)</bold> FCN. <bold>(E)</bold> PSPNet; <bold>(F)</bold> SegNet; <bold>(G)</bold> BiseNet; <bold>(H)</bold> ICNet; <bold>(I)</bold> DeepLabv3+; <bold>(J)</bold> Ours.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g011.tif">
<alt-text content-type="machine-generated">A series of ten images labeled A to J showing plants in different visual filters. Image A shows a natural view with green plants against soil. Images B to J display various segmentation techniques highlighting plants in distinct colors&#x2014;green for some leaves and red for others&#x2014;against a darkened soil background for enhanced visibility and analysis.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Ablation experiments</title>
<sec id="s3_4_1">
<label>3.4.1</label>
<title>Ablation experiments of the C-CFF module</title>
<p>To assess the effectiveness of the improved C-CFF module, we conducted ablation studies using a controlled variable approach. As shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, incorporating either CAM or SAM individually led to slight increases in parameters and FLOPs, yielding mIoU improvements of 0.53% and 0.65%, respectively. By contrast, integrating CBAM into the CFF module improved mIoU by 1.51% over the baseline, with only a modest increase of 0.7M parameters and 0.53 GFLOPs. Although the 7&#xd7;7 convolution in the spatial attention mechanism slightly reduced the inference speed, the overall computational overhead remained minimal. These findings validate the effectiveness of the C-CFF module in enhancing segmentation performance. However, the relatively limited improvements achieved by CAM and SAM individually warrant further analysis. A possible explanation is that, after multiple layers of convolution and pooling, the deep feature maps already contain abundant semantic information. In such scenarios, employing a single channel or spatial attention mechanism may fail to extract additional meaningful features, potentially resulting in redundancy and misaligned attention. Moreover, CAM and SAM were integrated independently into the CFF module without interaction, which limited their overall effectiveness. In contrast, CBAM sequentially applies channel and spatial attention, first highlighting informative channels and then emphasizing critical spatial regions. This sequential mechanism enhances feature selection more effectively. Experimental results demonstrate that CBAM yields more significant performance improvements than using CAM or SAM individually.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Results of ablation experiments with the C-CFF module.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Module</th>
<th valign="middle" align="center">mIoU (%)</th>
<th valign="middle" align="center">Parameters (M)</th>
<th valign="middle" align="center">FLOPs (G)</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">CFF</td>
<td valign="middle" align="center">84.06</td>
<td valign="middle" align="center">2.71</td>
<td valign="middle" align="center">14.83</td>
<td valign="middle" align="center">44.01</td>
</tr>
<tr>
<td valign="middle" align="center">CFF+CAM</td>
<td valign="middle" align="center">85.19</td>
<td valign="middle" align="center">2.77</td>
<td valign="middle" align="center">15.04</td>
<td valign="middle" align="center">43.36</td>
</tr>
<tr>
<td valign="middle" align="center">CFF+SAM</td>
<td valign="middle" align="center">84.71</td>
<td valign="middle" align="center">2.81</td>
<td valign="middle" align="center">15.16</td>
<td valign="middle" align="center">43.27</td>
</tr>
<tr>
<td valign="middle" align="center">CFF+CBAM</td>
<td valign="middle" align="center">85.57</td>
<td valign="middle" align="center">2.89</td>
<td valign="middle" align="center">15.326</td>
<td valign="middle" align="center">42.89</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_4_2">
<label>3.4.2</label>
<title>Ablation experiments of different modules</title>    <p>To evaluate the impact of the proposed improvements on DeepLabv3+ performance, we conducted ablation experiments using a self-constructed dataset. The baseline model was DeepLabv3+ with an Xception backbone. Four ablation settings were assessed using standard semantic segmentation metrics. <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> shows the experiment&#x2019;s results, which&#x221a; indicate that the specified module was employed.</p>
<list list-type="order">
<list-item>
<p>Group 1: The baseline model&#x2019;s backbone was replaced with the MobileNetV2 architecture, and the standard convolution operations in the encoder&#x2013;decoder were substituted with depthwise separable dilated convolutions.</p>
</list-item>
<list-item>
<p>Group 2: Building on Group 1, the S-ASPP structure was introduced, followed by the CBAM.</p>
</list-item>
<list-item>
<p>Group 3: Based on Group 1, the C-CFF module was incorporated into the decoder to fuse features across different scales.</p>
</list-item>
<list-item>
<p>Group 4: The C-CFF module was further integrated with Group 2.</p>
</list-item>
</list>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Results of ablation experiments with each module.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">MobileNetV2</th>
<th valign="middle" align="center">S-ASPP</th>
<th valign="middle" align="center">C-CFF</th>
<th valign="middle" align="center">mIoU (%)</th>
<th valign="middle" align="center">Parameters (M)</th>
<th valign="middle" align="center">FLOPs (G)</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">82.72</td>
<td valign="middle" align="center">54.714</td>
<td valign="middle" align="center">167.139</td>
<td valign="middle" align="center">18.4</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">83.94</td>
<td valign="middle" align="center">2.745</td>
<td valign="middle" align="center">13.145</td>
<td valign="middle" align="center">31.75</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">84.75</td>
<td valign="middle" align="center">2.791</td>
<td valign="middle" align="center">13.612</td>
<td valign="middle" align="center">33.27</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">85.43</td>
<td valign="middle" align="center">2.847</td>
<td valign="middle" align="center">15.326</td>
<td valign="middle" align="center">35.31</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">85.57</td>
<td valign="middle" align="center">2.890</td>
<td valign="middle" align="center">15.767</td>
<td valign="middle" align="center">42.89</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> presents the results of replacing the Xception backbone with MobileNetV2. This modification, combined with the use of depthwise separable dilated convolutions, improved the mIoU by 1.22%, reduced the number of parameters by 51.97M, decreased FLOPs by 153.99G, and increased the inference speed to 31.75 FPS. The integration of the S-ASPP module further improved segmentation performance, increasing the mIoU by an additional 1.49%, with only slight increases of 0.102M in parameters and 2.18GFLOPs. The introduction of the C-CFF structure further refined the model architecture and contributed to enhanced segmentation accuracy. When all three modules were combined, the model achieved an mIoU of 85.57%, representing a 2.85% improvement over the original configuration. The number of parameters was reduced to 2.89M, FLOPs were decreased to approximately one-tenth of the original value, and the inference speed nearly doubled. Each modification contributed to a more lightweight model design while simultaneously enhancing segmentation accuracy.</p>
</sec>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Testing on the PASCAL VOC 2012</title>
<p>The PASCAL VOC 2012 dataset is widely used in computer vision and includes 21 semantic classes, such as car, person, cat, and dog. It serves as a standard benchmark for evaluating semantic segmentation models. We evaluated the performance of the proposed DSC-DeepLabv3+ model on this dataset to assess its generalization capability. As illustrated in <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>, the model achieves competitive segmentation results. As detailed in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>, our model achieves higher mIoU compared to several existing methods, including U-Net, FCN, PSPNet, and BiSeNet. Although the mIoU is slightly lower than that of the original DeepLabv3+ with an Xception backbone, our model demonstrates significant advantages in terms of parameter count, computational cost, and inference speed. Specifically, compared with the MobileNetV2-based DeepLabv3+, our model improves mIoU by 1.72%, reduces the number of parameters by 50%, lowers computational cost by 73%, and increases inference speed by 11.88 FPS. Furthermore, compared to BiSeNet, our model achieves 3.69% higher mIoU, reduces parameters by 2.89M, and lowers FLOPs by 34.9G. Compared to ICNet, it achieves 3.42% higher mIoU, with a reduction of 23.61M in parameters and 12.9G in FLOPs. Although the inference speed is 6.05 FPS lower than that of BiSeNet, the proposed model achieves a favorable balance between segmentation accuracy and model efficiency. Overall, the experimental results demonstrate the strong generalization capability of the proposed model.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>The PASCAL VOC 2012 dataset&#x2019;s mIoU values for all categories.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1647736-g012.tif">
<alt-text content-type="machine-generated">Bar chart showing Intersection over Union (IoU) scores for various classes, with a mean IoU (mIoU) of 74.31 percent. Most classes have scores above 0.7, except for &#x201c;chair&#x201d; at 0.18, &#x201c;pottedplant&#x201d; at 0.46, and &#x201c;sofa&#x201d; at 0.56.</alt-text>
</graphic>
</fig>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Comparison of segmentation results of different models on PASCAL VOC 2012.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Models</th>
<th valign="middle" align="center">Backbone</th>
<th valign="middle" align="center">mIoU (%)</th>
<th valign="middle" align="center">Parameters (M)</th>
<th valign="middle" align="center">FLOPs (G)</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Unet</td>
<td valign="middle" align="center">VGG16</td>
<td valign="middle" align="center">58.78</td>
<td valign="middle" align="center">43.93</td>
<td valign="middle" align="center">184.4</td>
<td valign="middle" align="center">10.3</td>
</tr>
<tr>
<td valign="middle" align="center">PSPNet</td>
<td valign="middle" align="center">ResNet50</td>
<td valign="middle" align="center">68.94</td>
<td valign="middle" align="center">46.716</td>
<td valign="middle" align="center">118.47</td>
<td valign="middle" align="center">26.7</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="center">DeepLabv3+</td>
<td valign="middle" align="center">Xception</td>
<td valign="middle" align="center">
<bold>75.65</bold>
</td>
<td valign="middle" align="center">54.714</td>
<td valign="middle" align="center">167.139</td>
<td valign="middle" align="center">16.4</td>
</tr>
<tr>
<td valign="middle" align="center">MobileNetV2</td>
<td valign="middle" align="center">72.59</td>
<td valign="middle" align="center">5.81</td>
<td valign="middle" align="center">56.248</td>
<td valign="middle" align="center">21.68</td>
</tr>
<tr>
<td valign="middle" align="center">FCN</td>
<td valign="middle" align="center">VGG16</td>
<td valign="middle" align="center">71.5</td>
<td valign="middle" align="center">32.75</td>
<td valign="middle" align="center">89.8</td>
<td valign="middle" align="center">9.16</td>
</tr>
<tr>
<td valign="middle" align="center">BiSeNet</td>
<td valign="middle" align="center">Xception</td>
<td valign="middle" align="center">70.62</td>
<td valign="middle" align="center">5.8</td>
<td valign="middle" align="center">50.3</td>
<td valign="middle" align="center">
<bold>39.61</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">ICNet</td>
<td valign="middle" align="center">ResNet50</td>
<td valign="middle" align="center">70.89</td>
<td valign="middle" align="center">26.5</td>
<td valign="middle" align="center">28.3</td>
<td valign="middle" align="center">28.5</td>
</tr>
<tr>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">MobileNetV2</td>
<td valign="middle" align="center">74.31</td>
<td valign="middle" align="center">
<bold>2.89</bold>
</td>
<td valign="middle" align="center">
<bold>15.326</bold>
</td>
<td valign="middle" align="center">33.56</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Ablation study results. Bold values indicate the best results.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>Semantic segmentation models have found increasing application in agriculture, where large-scale architectures have demonstrated notable performance improvements. However, deploying models with large parameter counts on embedded devices such as agricultural robots and drones remains challenging due to limited computational and memory resources. Although lightweight models reduce parameter counts, they frequently suffer from performance degradation, especially in complex field conditions characterized by high misclassification rates. To overcome these limitations, this study presents DSC-DeepLabv3+, an improved lightweight semantic segmentation model built upon the DeepLabv3+ framework. Specifically, the original backbone is replaced, and standard convolutions in both the ASPP module and decoder are substituted with depthwise separable convolutions, reducing computational complexity. A strip pooling mechanism is incorporated into the ASPP module, forming an S-ASPP structure that enhances the model&#x2019;s capacity to capture multi-scale contextual information. Furthermore, integrating the CBAM module suppresses background interference and strengthens feature representation. In the decoder, the improved C-CFF module facilitates the efficient integration of multi-stage features, thereby reducing pixel-level information loss and enhancing prediction accuracy. To evaluate the proposed model, a corn&#x2013;weed segmentation dataset was constructed. Experimental results demonstrate that DSC-DeepLabv3+ achieves an mIoU of 85.57% and an inference speed of 42.89 FPS, with only 2.89M parameters and 15.326 GFLOPs. Although its mIoU is only slightly lower (by 0.71%) than that of the original DeepLabv3+, the proposed model significantly reduces model size and computational overhead. Moreover, it outperforms lightweight baseline models such as BiSeNet under resource-constrained conditions, demonstrating superior efficiency and accuracy. Its generalization capability is further confirmed through evaluation on the PASCAL VOC 2012 dataset.</p>
<p>Despite the encouraging results achieved in this study, several limitations remain. One notable issue is the absence of direct comparisons with recently proposed state-of-the-art lightweight models specifically designed for agricultural scenarios, such as the improved U-Net (<xref ref-type="bibr" rid="B43">Zuo and Li, 2024</xref>) and DFFANet (<xref ref-type="bibr" rid="B9">Feng et&#xa0;al., 2022</xref>). Although these models are well-recognized in the field and were considered for inclusion, reliable reproduction was impeded due to the lack of publicly available source code and insufficient hyperparameter details in their original publications. Attempts to contact the corresponding authors were unsuccessful. While our experiments included several widely adopted and representative baseline models, the omission of the most recent architectures may limit the completeness of performance evaluation and the positioning of our approach within the current research landscape. In future work, we aim to include such models once reliable implementations become accessible. Another limitation lies in the model&#x2019;s robustness under complex and variable environmental conditions, such as lighting changes, cluttered backgrounds, and diverse weed morphologies. Although data augmentation was employed to simulate some of these scenarios, real-world agricultural environments are often more unpredictable, featuring strong shadows, overlapping vegetation, and high similarity between foreground and background. These factors may challenge the model&#x2019;s generalization capability. Furthermore, the current dataset may not adequately capture the full diversity of weed species across different geographic regions and growth stages. To address these issues, future research will focus on expanding the dataset to include more representative field conditions and broader weed categories. Enhancing model robustness through adaptive attention mechanisms, domain generalization techniques, or the integration of multispectral and temporal data will also be explored. Additionally, further optimization of the lightweight architecture will be pursued to support real-time deployment on resource-constrained agricultural platforms, ultimately advancing its applicability in precision agriculture.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions</title>
<p>To tackle the challenge of efficient weed identification in maize fields under resource-constrained conditions, this study offers the following key contributions. A maize field weed image dataset was constructed and preprocessed to reflect a wide range of realistic growth conditions. Subsequently, a novel lightweight semantic segmentation model, termed DSC-DeepLabv3+, was proposed. The model maintains a compact architecture, requiring merely 2.89M parameters and 15.236 GFLOPs, thereby addressing memory and processing limitations commonly encountered in field-level deployment. Experimental evaluations show that DSC-DeepLabv3+ achieves an mIoU of 85.57% on the constructed maize weed dataset, surpassing both conventional and lightweight benchmark models. Future work will focus on extending the model to other crop types to enhance its generalizability and contribute to the advancement of precision agriculture.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>HF: Conceptualization, Software, Writing &#x2013; review &amp; editing. XL: Conceptualization, Resources, Software, Visualization, Writing &#x2013; original draft. LZ: Funding acquisition, Investigation, Methodology, Supervision, Writing &#x2013; review &amp; editing. PX: Investigation, Resources, Writing &#x2013; review &amp; editing. TW: Data curation, Methodology, Validation, Writing &#x2013; original draft. WL: Investigation, Validation, Writing &#x2013; original draft. YF: Conceptualization, Project administration, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This research was supported by the Jilin Provincial Department of Education Science and Technology Research Project (JJKH20250567KJ).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Agarwal</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hariharan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Rao</surname> <given-names>M. N.</given-names>
</name>
<name>
<surname>Agarwal</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>) &#x201c;<article-title>Weed identification using K-means clustering with color spaces features in multi-spectral images taken by UAV</article-title>,&#x201d; in <conf-name>2021 IEEE International Geoscience and Remote Sensing Symposium IGARSS</conf-name>. <fpage>7047</fpage>&#x2013;<lpage>7050</lpage> (<publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Badrinarayanan</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Kendall</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Cipolla</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>SegNet: A deep convolutional encoder-decoder architecture for image segmentation</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>39</volume>, <fpage>2481</fpage>&#x2013;<lpage>2495</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tpami.2016.2644615</pub-id>, PMID: <pub-id pub-id-type="pmid">28060704</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Papandreou</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Kokkinos</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Murphy</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Yuille</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Semantic image segmentation with deep convolutional nets and fully connected CRFs</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1412.7062</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Papandreou</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Kokkinos</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Murphy</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Yuille</surname> <given-names>A. L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DeepLab: semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected CRFs</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>40</volume>, <fpage>834</fpage>&#x2013;<lpage>848</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tpami.2017.2699184</pub-id>, PMID: <pub-id pub-id-type="pmid">28463186</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Papandreou</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Schroff</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Adam</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2016</year>) &#x201c;<article-title>Encoder-decoder with atrous separable convolution for semantic image segmentation</article-title>,&#x201d; in <conf-name>Proceedings of the European conference on computer vision</conf-name>. <fpage>801</fpage>&#x2013;<lpage>818</lpage> (<publisher-name>ECCV</publisher-name>).</citation></ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chollet</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2017</year>) &#x201c;<article-title>Xception: Deep learning with depthwise separable convolutions</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>1800</fpage>&#x2013;<lpage>1807</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2017.195.</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Espejo-Garcia</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Mylonas</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Athanasakos</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Fountas</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Vasilakoglou</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Towards weeds identification assistance through transfer learning</article-title>. <source>Comput. Electron. Agric.</source> <volume>171</volume>, <elocation-id>10</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105306</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chai</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>) &#x201c;<article-title>Rethinking bisenet for real-time semantic segmentation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <fpage>9716</fpage>&#x2013;<lpage>9725</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.00959</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname> <given-names>C. G.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>M. L.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>L. G.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Y. L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A lightweight real-time rice blast disease segmentation method based on DFFANet</article-title>. <source>Agriculture-Basel</source> <volume>12</volume>, <elocation-id>12</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture12101543</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fuentes-Pacheco</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Torres-Olivares</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Roman-Rangel</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Cervantes</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Juarez-Lopez</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Hermosillo-Valadez</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Fig plant segmentation from aerial images using a deep convolutional encoder-decoder network</article-title>. <source>Remote Sens.</source> <volume>11</volume>, <fpage>18</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs11101157</pub-id>
</citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamuda</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Glavin</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A survey of image processing techniques for plant extraction and segmentation in the field</article-title>. <source>Comput. Electron. Agric.</source> <volume>125</volume>, <fpage>184</fpage>&#x2013;<lpage>199</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2016.04.024</pub-id>
</citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K. M.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X. Y.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S. Q.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Spatial pyramid pooling in deep convolutional networks for visual recognition</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>37</volume>, <fpage>1904</fpage>&#x2013;<lpage>1916</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tpami.2015.2389824</pub-id>, PMID: <pub-id pub-id-type="pmid">26353135</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hou</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>M.-M.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>) &#x201c;<article-title>Strip pooling: Rethinking spatial pooling for scene parsing</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <fpage>4003</fpage>&#x2013;<lpage>4012</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR42600.2020.00406</pub-id>
</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Janneh</surname> <given-names>L. L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y. J.</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>Z. W.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Y. T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Multi-level feature re-weighted fusion for the semantic segmentation of crops and weeds</article-title>. <source>J. King Saud University-Computer Inf. Sci.</source> <volume>35</volume>, <fpage>13</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jksuci.2023.03.023</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y. Q.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X. K.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H. B.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W. M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>EACNet: enhanced asymmetric convolution for real-time semantic segmentation</article-title>. <source>IEEE Signal Process. Lett.</source> <volume>28</volume>, <fpage>234</fpage>&#x2013;<lpage>238</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/lsp.2021.3051845</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Yun</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Dabnet: Depth-wise asymmetric bottleneck for real-time semantic segmentation</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1907.11357</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moreau</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Busset</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Matejicek</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Prudent</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Colbach</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Water limitation affects weed competitive ability for light. A demonstration using a model-based approach combined with an automated watering platform</article-title>. <source>Weed Res.</source> <volume>62</volume>, <fpage>381</fpage>&#x2013;<lpage>392</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/wre.12554</pub-id>
</citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muola</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Fuchs</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Laihonen</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rainio</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Heikkonen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Ruuskanen</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Risk in the circular food economy: Glyphosate-based herbicide residues in manure fertilizers decrease crop yield</article-title>. <source>Sci. Total Environ.</source> <volume>750</volume>, <fpage>7</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scitotenv.2020.141422</pub-id>, PMID: <pub-id pub-id-type="pmid">32858290</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname> <given-names>H. H.</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>Y. D.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>W. C.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>Y. S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep dual-resolution networks for real-time and accurate semantic segmentation of traffic scenes</article-title>. <source>IEEE Trans. Intelligent Transportation Syst.</source> <volume>24</volume>, <fpage>3448</fpage>&#x2013;<lpage>3460</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tits.2022.3228042</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paszke</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chaurasia</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Culurciello</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Enet: A deep neural network architecture for real-time semantic segmentation</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1606.02147</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Romera</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Alvarez</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Bergasa</surname> <given-names>L. M.</given-names>
</name>
<name>
<surname>Arroyo</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>ERFNet: efficient residual factorized convNet for real-time semantic segmentation</article-title>. <source>IEEE Trans. Intelligent Transportation Syst.</source> <volume>19</volume>, <fpage>263</fpage>&#x2013;<lpage>272</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tits.2017.2750080</pub-id>
</citation></ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ronneberger</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Fischer</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Brox</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>) &#x201c;<article-title>U-net: Convolutional networks for biomedical image segmentation</article-title>,&#x201d; in <conf-name>Medical image computing and computer-assisted intervention&#x2013;MICCAI 2015: 18th international conference</conf-name>, <conf-loc>Munich, Germany</conf-loc>, <conf-date>October 5-9, 2015</conf-date>.  <volume>9351</volume>, <fpage>234</fpage>&#x2013;<lpage>241</lpage> (<publisher-loc>Cham.</publisher-loc>: <publisher-name>Springer</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-24574-4_28</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Russell</surname> <given-names>B. C.</given-names>
</name>
<name>
<surname>Torralba</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Murphy</surname> <given-names>K. P.</given-names>
</name>
<name>
<surname>Freeman</surname> <given-names>W. T.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>LabelMe: A database and web-based tool for image annotation</article-title>. <source>Int. J. Comput. Vision</source> <volume>77</volume>, <fpage>157</fpage>&#x2013;<lpage>173</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11263-007-0090-8</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sandler</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Howard</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhmoginov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>L.-C.</given-names>
</name>
</person-group> (<year>2018</year>) &#x201c;<article-title>Mobilenetv2: Inverted residuals and linear bottlenecks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>4510</fpage>&#x2013;<lpage>4520</lpage>.</citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shah</surname> <given-names>T. M.</given-names>
</name>
<name>
<surname>Nasika</surname> <given-names>D. P. B.</given-names>
</name>
<name>
<surname>Otterpohl</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Plant and weed identifier robot as an agroecological tool using artificial neural networks for image identification</article-title>. <source>Agriculture-Basel</source> <volume>11</volume>, <fpage>31</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture11030222</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shelhamer</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Darrell</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Fully convolutional networks for semantic segmentation</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>39</volume>, <fpage>640</fpage>&#x2013;<lpage>651</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tpami.2016.2572683</pub-id>, PMID: <pub-id pub-id-type="pmid">27244717</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Yi</surname> <given-names>Q. M.</given-names>
</name>
<name>
<surname>Weng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Z. K.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>A. W.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>LMFFNet: A well-balanced lightweight network for fast and accurate semantic segmentation</article-title>. <source>IEEE Trans. Neural Networks Learn. Syst.</source> <volume>34</volume>, <fpage>3205</fpage>&#x2013;<lpage>3219</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tnnls.2022.3176493</pub-id>, PMID: <pub-id pub-id-type="pmid">35622806</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Subeesh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Bhole</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Chandel</surname> <given-names>N. S.</given-names>
</name>
<name>
<surname>Rajwade</surname> <given-names>Y. A.</given-names>
</name>
<name>
<surname>Rao</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Deep convolutional neural network models for weed detection in polyhouse grown bell peppers</article-title>. <source>Artif. Intell. Agric.</source> <volume>6</volume>, <fpage>47</fpage>&#x2013;<lpage>54</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aiia.2022.01.002</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>H. N.</given-names>
</name>
<name>
<surname>He</surname> <given-names>X. H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H. F.</given-names>
</name>
<name>
<surname>Kong</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>X. J.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Adaptive sparse lightweight multi-scale hybrid network for remote sensing image semantic segmentation</article-title>. <source>Expert Syst. Appl.</source> <volume>280</volume>, <fpage>17</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2025.127347</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>) &#x201c;<article-title>Lednet: A lightweight encoder-decoder network for real-time semantic segmentation</article-title>,&#x201d; in <conf-name>2019 IEEE international conference on image processing (ICIP)</conf-name>. <fpage>1860</fpage>&#x2013;<lpage>1864</lpage> (<publisher-name>IEEE</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICIP.2019.8803154</pub-id>
</citation></ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J.-Y.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Cbam: Convolutional block attention module</article-title>,&#x201d; in <conf-name>Proceedings of the European conference on computer vision</conf-name>. <fpage>3</fpage>&#x2013;<lpage>19</lpage> (<publisher-loc>Springer, Cham</publisher-loc>: <publisher-name>ECCV</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-01234-2_1</pub-id>
</citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>T. Y.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y. D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>CGNet: A light-weight context guided network for semantic segmentation</article-title>. <source>IEEE Trans. Image Process.</source> <volume>30</volume>, <fpage>1169</fpage>&#x2013;<lpage>1179</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tip.2020.3042065</pub-id>, PMID: <pub-id pub-id-type="pmid">33306466</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>X. J.</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Y. F.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S. Y.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>H. L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A novel feature selection strategy based on salp swarm algorithm for plant disease detection</article-title>. <source>Plant Phenomics</source> <volume>2023</volume>, <fpage>17</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/plantphenomics.0039</pub-id>, PMID: <pub-id pub-id-type="pmid">37228513</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Bhattacharyya</surname> <given-names>S. P.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>PIDNet: A real-time semantic segmentation network inspired by PID controllers</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <fpage>19529</fpage>&#x2013;<lpage>19539</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR52729.2023.01871</pub-id>
</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>Z. Y.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>C. Q.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>X. M.</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>Y. L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H. S.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>A lightweight semantic segmentation method for concrete bridge surface diseases based on improved DeeplabV3+</article-title>. <source>Sci. Rep.</source> <volume>15</volume>, <fpage>12</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-025-95518-5</pub-id>, PMID: <pub-id pub-id-type="pmid">40133624</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>C. Q.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>C. X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J. B.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>C. H.</given-names>
</name>
<name>
<surname>Sang</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>BiSeNet V2: bilateral network with guided aggregation for real-time semantic segmentation</article-title>. <source>Int. J. Comput. Vision</source> <volume>129</volume>, <fpage>3051</fpage>&#x2013;<lpage>3068</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11263-021-01515-2</pub-id>
</citation></ref>
<ref id="B37">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Sang</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>) &#x201c;<article-title>Bisenet: Bilateral segmentation network for real-time semantic segmentation</article-title>,&#x201d; in <conf-name>Proceedings of the European conference on computer vision</conf-name>. <fpage>325</fpage>&#x2013;<lpage>341</lpage> (<publisher-loc>Cham.</publisher-loc>: <publisher-name>ECCV</publisher-name>).</citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>X. L.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>B. C.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Z. Y.</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>T. B.</given-names>
</name>
</person-group> (<year>2022</year>b). <article-title>LAANet: lightweight attention-guided asymmetric network for real-time semantic segmentation</article-title>. <source>Neural Computing Appl.</source> <volume>34</volume>, <fpage>3573</fpage>&#x2013;<lpage>3587</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00521-022-06932-z</pub-id>
</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C. Y.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>a). <article-title>Segmentation algorithm for overlap recognition of seedling lettuce and weeds based on SVM and image blocking</article-title>. <source>Comput. Electron. Agric.</source> <volume>201</volume>, <elocation-id>10</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.107284</pub-id>
</citation></ref>
<ref id="B40">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>ICNet for real-time semantic segmentation on high-resolution images</article-title>. In <person-group person-group-type="editor">
<name>
<surname>Ferrari</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Hebert</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sminchisescu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Weiss</surname> <given-names>Y.</given-names>
</name>
</person-group> (Eds.), <source>Computer Vision &#x2013; ECCV 2018 <italic>(Lecture Notes in Computer Science)</italic>
</source>.  Vol. <volume>11207</volume>. (<publisher-name>Springer, Cham</publisher-name>), <fpage>418</fpage>&#x2013;<lpage>434</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-01219-9_25</pub-id>
</citation></ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>) <article-title>Pyramid scene parsing network</article-title>. In <conf-name>2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <fpage>6230</fpage>&#x2013;<lpage>6239</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2017.660</pub-id>
</citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname> <given-names>K. L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C. L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A field weed density evaluation method based on UAV imaging and modified U-net</article-title>. <source>Remote Sens.</source> <volume>13</volume>, <fpage>19</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs13020310</pub-id>
</citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zuo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W. W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An improved UNet lightweight network for semantic segmentation of weed images in corn fields</article-title>. <source>Cmc-Computers Materials Continua</source> <volume>79</volume>, <fpage>4413</fpage>&#x2013;<lpage>4431</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.32604/cmc.2024.049805</pub-id>
</citation></ref>
</ref-list>
</back>
</article>