<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1506524</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>PD-YOLO: a novel weed detection method based on multi-scale feature fusion</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Shengzhou</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2886393/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Zihan</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xie</surname>
<given-names>Jialong</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Hewei</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2849633/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Guo</surname>
<given-names>Jianwen</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2235120/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>School of Mechanical Engineering, Dongguan University of Technology</institution>, <addr-line>Dongguan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Qingxia (Jenny) Wang, University of Southern Queensland, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Elio Romano, Centro di Ricerca per l&#x2019;Ingegneria e le Trasformazioni Agroalimentari (CREA-IT), Italy</p>
<p>Yalin Wu, Peking University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Jianwen Guo, <email xlink:href="mailto:guojw@dgut.edu.cn">guojw@dgut.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>04</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1506524</elocation-id>
<history>
<date date-type="received">
<day>05</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>03</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Li, Chen, Xie, Zhang and Guo</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Li, Chen, Xie, Zhang and Guo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The deployment of robots for automated weeding holds significant promise in promoting sustainable agriculture and reducing labor requirements, with vision based detection being crucial for accurate weed identification. However, weed detection through computer vision presents various challenges, such as morphological similarities between weeds and crops, large-scale variations, occlusions, and the small size of the target objects.</p>
</sec>
<sec>
<title>Methods</title>
<p>To overcome these challenges, this paper proposes a novel object detection model, PD-YOLO, based on multi-scale feature fusion. Building on the YOLOv8n framework, the model introduces a Parallel Focusing Feature Pyramid (PF-FPN), which incorporates two key components: the Feature Filtering and Aggregation Module (FFAM) and the Hierarchical Adaptive Recalibration Fusion Module (HARFM). These modules facilitate efficient feature fusion both laterally and radially across the network. Furthermore, the inclusion of a dynamic detection head (Dyhead) significantly enhances the model&#x2019;s capacity to detect and locate weeds in complex environments.</p>
</sec>
<sec>
<title>Results and discussion</title>
<p>Experimental results on two public weed datasets demonstrate the superior performance of PD-YOLO over state-the-art models, with a modest increase in computational cost. PD-YOLO improves the mean average precision (mAP) by 1.7% and 1.8% on the CottonWeedDet12 dataset at thresholds of 0.5 and 0.5-0.95, respectively. This research not only presents an efficient and accurate weed detection method but also offers new insights and technological advances for automated weed detection in agriculture.</p>
</sec>
</abstract>
<kwd-group>
<kwd>weed detection</kwd>
<kwd>object detection</kwd>
<kwd>YOLO</kwd>
<kwd>multi-scale feature fusion</kwd>
<kwd>dynamic detection head</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="7"/>
<equation-count count="22"/>
<ref-count count="54"/>
<page-count count="20"/>
<word-count count="10197"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Weeds are one of the major factors affecting agriculture. Currently, the damage caused by weeds to agriculture reaches as high as 34% (<xref ref-type="bibr" rid="B22">Oerke, 2006</xref>). For decades, herbicides have been widely adopted as the preferred method for weed management in global agriculture; however, herbicides often have adverse environmental side effects (<xref ref-type="bibr" rid="B16">Kudsk and Streibig, 2003</xref>). With the development of precision agriculture and the widespread use of intelligent agricultural machinery, robots for weed control hold great potential in building environmentally friendly agriculture and reducing labor demands (<xref ref-type="bibr" rid="B46">You et&#xa0;al., 2020</xref>). Intelligent robots rely on real-time weed detection systems with high accuracy to locate weeds, but weed detection in actual farmland environments still faces the following challenges.</p>
<p>Firstly, the coexistence of inter-class phenotypic similarity and intra-class morphological variation poses significant challenges. Weeds and crops exhibit substantial overlap in color features, particularly during the seedling stage, with notable similarities in leaf shape and texture characteristics during growth (<xref ref-type="bibr" rid="B12">Hasan et&#xa0;al., 2021</xref>). For instance, Italian ryegrass (Lolium multiflorum) and wheat are visually indistinguishable without expert knowledge (<xref ref-type="bibr" rid="B2">Bansal et&#xa0;al., 2024</xref>), rendering traditional detection methods ineffective. Compounding this complexity, dynamic variations in leaf color, morphology, and texture occur across different growth stages of the same weed species (<xref ref-type="bibr" rid="B34">Veeragandham and Santhi, 2021</xref>), hindering the establishment of stable feature representations in detection models. Studies have shown that YOLOX and YOLOv8 models experienced accuracy declines of 14.5% and 14.2%, respectively, in identifying eight cross-season weed categories common in cotton fields. This degradation primarily stems from seasonal variations in lighting, background conditions, and weed growth states (<xref ref-type="bibr" rid="B10">Deng et&#xa0;al., 2024</xref>).</p>
<p>Secondly, the sheer diversity of weed species (<xref ref-type="bibr" rid="B6">Chen et&#xa0;al., 2024</xref>), combined with the simultaneous presence of weeds at varying developmental stages within agroecosystems, creates scale differences spanning three orders of magnitude. Empirical analysis using the Weed25 dataset revealed significant disparities in recognition performance: Asiatic smartweed (Polygonum aviculare) achieved a mean average precision (mAP) of 62.92%, while velvetleaf (Abutilon theophrasti) reached 99.70% (<xref ref-type="bibr" rid="B39">Wang et&#xa0;al., 2022</xref>). Such scale heterogeneity complicates the development of algorithms capable of effectively detecting multi-category weeds. Furthermore, leaf overlap and weed occlusion in dense scenarios exacerbate differentiation and detection challenges (<xref ref-type="bibr" rid="B40">Wang et&#xa0;al., 2019</xref>). In cabbage fields, weed detection not only struggles with color similarity but also contends with illumination variations and leaf occlusion, leading to suboptimal performance in direct detection methods addressing these issues (<xref ref-type="bibr" rid="B30">Sun et&#xa0;al., 2024</xref>).</p>
<p>Finally, the high-density distribution and small size of weed targets frequently result in missed detections and false positives. <xref ref-type="bibr" rid="B42">Wu et&#xa0;al. (2023)</xref>. proposed an enhanced YOLO-V4 model tailored for small weed detection in farmland, improving the mAP by 4.2%. However, despite advancements in lightweight performance, the parameter count of model remains high at 42.54 million, posing deployment challenges.</p>
<p>To address the aforementioned challenges, this paper proposes an PD-YOLO method based on a multi-scale feature fusion network, building on YOLOv8n. The method introduces an innovative Parallel Focusing Feature Pyramid (PF-FPN), which effectively improves the accurate classification and localization of different types of weeds in complex environments. The PF-FPN includes the Feature Filtering and Aggregation Module (FFAM) and the Hierarchical Adaptive Recalibration Fusion Module (HARFM). The FFAM module utilizes deep convolution and attention mechanisms to preliminarily filter and extract features, adaptively adjusting and fusing multi-scale features to capture rich semantic information. This enhances small object detection and multi-scale feature fusion, thereby improving the model&#x2019;s accuracy and robustness. The HARFM module leverages attention mechanisms to fuse features at different levels, achieving adaptive optimization and enhancing the model&#x2019;s expressive power. To further improve detection performance, a dynamic detection head (Dyhead) (<xref ref-type="bibr" rid="B8">Dai et&#xa0;al., 2021</xref>) is introduced, enhancing the model&#x2019;s stability and accuracy in complex backgrounds. Ultimately, the proposed PD-YOLO model integrates the PF-FPN network and Dyhead architecture, using YOLOv8n as the base framework. This optimization of feature fusion and model representation capabilities significantly enhances the overall accuracy of weed detection.</p>
<p>The remainder of this paper is organized as follows: Section 2 introduces related research work; Section 3 presents the PD-YOLO model; Section 4 describes the experiments conducted on the model; Section 5 provides relevant discussions; and Section 6 concludes the paper.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related works</title>
<p>Early research in weed detection primarily relied on traditional image processing techniques (<xref ref-type="bibr" rid="B41">Wu et&#xa0;al., 2021</xref>). These methods involved extracting features such as color, texture, and shape from images, which were then used in conjunction with machine learning algorithms like Random Forests or Support Vector Machines for weed identification (<xref ref-type="bibr" rid="B26">Sabzi et&#xa0;al., 2020</xref>). For example, Islam et&#xa0;al (<xref ref-type="bibr" rid="B15">Islam et&#xa0;al., 2021</xref>). achieved efficient weed detection in Unmanned Aerial Vehicle (UAV) imagery through image orthorectification combined with machine learning algorithms, reaching accuracy rates of 98.40% on the original dataset and 94.72% on an extended dataset. The success of these techniques heavily depended on the quality of image acquisition, preprocessing, and feature extraction, as these factors directly influenced the performance and generalization ability of the algorithms.</p>
<p>With the advent of deep learning, object detection methods have revolutionized weed detection due to their superior efficiency and accuracy. <xref ref-type="bibr" rid="B33">Tang et&#xa0;al. (2017)</xref>. pioneered the use of K-means unsupervised feature learning in conjunction with Convolutional Neural Networks (CNNs), improving the identification accuracy of soybean seedlings and associated weeds to 92.89% through fine-tuning optimization. This approach effectively addressed the issues of instability and limited generalization found in manually designed feature-based methods. Similarly, dos Santos Ferreira et&#xa0;al (<xref ref-type="bibr" rid="B33">Tang et&#xa0;al., 2017</xref>) applied a ConvNets network to detect weeds in soybean crop images, classifying them into grass and broadleaf categories. This categorization enabled the targeted application of specific herbicides, achieving over 98% identification accuracy.</p>
<p>Currently, classical deep learning-based object detection methods are mainly categorized into single-stage and two-stage detection algorithms. Single-stage algorithms directly use neural networks to extract features from images and perform detection. Representative algorithms include SSD (<xref ref-type="bibr" rid="B18">Liu et&#xa0;al., 2016</xref>), the YOLO series (<xref ref-type="bibr" rid="B23">Redmon et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2021</xref>), and RT-DETR (<xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2021</xref>). Single-stage algorithms generally have the advantage of faster speed and better real-time detection performance, but they often suffer from lower localization accuracy. In contrast, two-stage algorithms first generate candidate regions and then classify and localize these regions. They usually achieve higher detection accuracy but are relatively slower and require more computational resources, with Faster R-CNN (<xref ref-type="bibr" rid="B25">Ren et&#xa0;al., 2015</xref>) being a representative example.</p>
<p>In weed detection, two-stage algorithms first generate potential weed-containing regions using object detection algorithms, followed by deep learning model classification to distinguish weeds from crops. Veeranampalayam Sivakumar et&#xa0;al (<xref ref-type="bibr" rid="B35">Veeranampalayam Sivakumar et&#xa0;al., 2020</xref>). constructed Faster R-CNN and SSD models and evaluated their performance for weed detection in soybean fields using UAV images. The results showed that both models performed well in weed detection, but Faster R-CNN outperformed SSD in terms of performance. This approach typically improves detection accuracy and reduces false positive rates, but its high computational complexity makes it difficult to meet real-time requirements.</p>
<p>To address the challenges of morphological diversity, scale variations, and complex backgrounds in farmland weed detection, researchers have proposed various improvements based on the YOLO series models. These methods enhance model performance in specific agricultural scenarios through strategies such as integrating attention mechanisms, optimizing multi-scale feature fusion, and improving small-target detection capabilities. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> systematically compares representative models in terms of improvement methods, parameter counts, computational costs, and performance metrics.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Systematic comparison of YOLO-based improvement methods and performance metrics for farmland weed detection tasks.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Application Scenario</th>
<th valign="middle" align="center">Improvement Methods</th>
<th valign="middle" align="center">Parameters (M)</th>
<th valign="middle" align="center">GFLOPS (G)</th>
<th valign="middle" align="center">mAP 0.5 (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">GTCBS-YOLOv5s (<xref ref-type="bibr" rid="B29">Shao et&#xa0;al., 2023</xref>)</td>
<td valign="middle" align="center">Rice field weed identification</td>
<td valign="middle" align="center">YOLOv5 + Ghost + C3Trans + CBAM + BiFPN + SIoU loss</td>
<td valign="middle" align="center">4.63</td>
<td valign="middle" align="center">25.1</td>
<td valign="middle" align="center">91.1</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOV7-G (<xref ref-type="bibr" rid="B47">Yu et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">Sesame field weed identification</td>
<td valign="middle" align="center">YOLOv7 + SimAM + C3 module + SPPFCSPC + Focal-SIoU loss</td>
<td valign="middle" align="center">0.57</td>
<td valign="middle" align="center">0.48</td>
<td valign="middle" align="center">56.6</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Riny (<xref ref-type="bibr" rid="B43">Xu et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">Herbal field weed tracking</td>
<td valign="middle" align="center">YOLOv7-tiny + FasterNet backbone + lightweight upsampling + ByteTrack</td>
<td valign="middle" align="center">10.1</td>
<td valign="middle" align="center">11.2</td>
<td valign="middle" align="center">91.7</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8-DMAS (<xref ref-type="bibr" rid="B53">Zheng et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">Cotton field weed detection</td>
<td valign="middle" align="center">YOLOv8 + DWR + MSBlock + small-target layer + ASFF + SoftNMS</td>
<td valign="middle" align="center">19.03</td>
<td valign="middle" align="center">51.2</td>
<td valign="middle" align="center">95.5</td>
</tr>
<tr>
<td valign="middle" align="center">RMS-DETR (<xref ref-type="bibr" rid="B11">Guo et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="center">Rice field weed identification</td>
<td valign="middle" align="center">DETR + multi-scale feature fusion + global context modeling</td>
<td valign="middle" align="center">40.8</td>
<td valign="middle" align="center">187</td>
<td valign="middle" align="center">85.1</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-CWD (<xref ref-type="bibr" rid="B21">Ma et&#xa0;al., 2025</xref>)</td>
<td valign="middle" align="center">Corn-weed detection</td>
<td valign="middle" align="center">YOLOv8 + hybrid attention + PIoU loss</td>
<td valign="middle" align="center">3.49</td>
<td valign="middle" align="center">9.6</td>
<td valign="middle" align="center">75.1</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Current YOLO-based farmland weed detection models generally suffer from scenario limitations and methodological homogenization. Most models, such as GTCBS-YOLOv5s (<xref ref-type="bibr" rid="B29">Shao et&#xa0;al., 2023</xref>) and YOLOv8-DMAS (<xref ref-type="bibr" rid="B53">Zheng et&#xa0;al., 2024</xref>), are optimized solely for single environments like rice or cotton fields, relying on repetitive technical approaches including attention mechanisms, multi-scale feature fusion, and loss function improvements, while lacking differentiated designs for weed morphological diversity and crop coexistence scenarios. Some models like RMS-DETR (<xref ref-type="bibr" rid="B11">Guo et&#xa0;al., 2024</xref>) with 40.8M parameters and 187 GFLOPs computational cost achieve only 85.1% accuracy, showing significant efficiency-accuracy imbalance. Lightweight models such as YOLOV7-G (<xref ref-type="bibr" rid="B47">Yu et&#xa0;al., 2024</xref>), though compressed to 0.57M parameters, suffer from high missed detection rates resulting in mAP as low as 56.6%, limiting practical applicability. Although a few models like YOLO-Riny (<xref ref-type="bibr" rid="B43">Xu et&#xa0;al., 2024</xref>) achieve edge device compatibility through structural lightweighting, their improvements remain confined to specific weed types Corydalis edulis and Setaria viridis in cornfields, failing to address complex multi-target interaction detection needs. While YOLO-CWD (<xref ref-type="bibr" rid="B21">Ma et&#xa0;al., 2025</xref>) achieves lightweight design with 75.1% mAP@50 and 9.6 GFLOPS in cornfield weed detection through hybrid attention mechanisms and PIoU loss function, its detection accuracy and model compactness still require further optimization. Existing studies generally lack cross-scenario generalization validation, showing weak support for multi-category weed interaction detection, environmental robustness, and crop-weed coexistence mechanisms, which constrains practical agricultural applications.</p>
<p>Existing weed detection methods based on general deep learning architectures face challenges due to the diversity in weed morphology and environmental conditions, making algorithm development for different plant species difficult (<xref ref-type="bibr" rid="B14">Hu et&#xa0;al., 2024</xref>). Additionally, Convolutional Neural Networks (CNNs), while extracting image features, are limited by the local receptive field of convolutional operators, making it hard to capture global information, which affects accurate image localization and classification (<xref ref-type="bibr" rid="B20">Luo et&#xa0;al., 2016</xref>). To overcome this limitation, researchers often employ multi-scale feature fusion techniques, with parallel multi-branch networks and serial skip-connection structures being two commonly used approaches.</p>
<p>In the Inception module of GoogLeNet, the parallel multi-branch network extracts multi-scale and hierarchical feature information from the same feature map using convolutional kernels of different sizes (<xref ref-type="bibr" rid="B31">Szegedy et&#xa0;al., 2015</xref>). Although this method takes advantage of convolution kernels with different receptive fields and carefully designed modules to learn rich multi-scale features, it overlooks semantic differences between features of different scales, which can lead to the loss of semantic information. High-level feature maps generated by the backbone network contain rich semantic information but lack the detailed information of objects, while low-level features, although containing precise object locations, lack sufficient semantic information.</p>
<p>To address this issue, high-dimensional features are typically upsampled and aligned with downsampled low-dimensional features, followed by pixel-wise summation to enhance semantic information. However, this method does not perform feature selection, merely summing pixel values across multiple feature layers, which may lead to redundant and repetitive information. Consequently, this approach fails to fully integrate and utilize diverse features (<xref ref-type="bibr" rid="B7">Chen et&#xa0;al., 2024</xref>).</p>
<p>In addition, feature fusion networks represent an efficient method for multi-scale fusion. The classic Feature Pyramid Network (FPN) (<xref ref-type="bibr" rid="B17">Lin et&#xa0;al., 2017</xref>) employs a top-down pyramid structure to achieve multi-scale feature fusion. However, due to its structural characteristics, FPN lacks sufficient high-level semantic information. To enhance local localization information, PANet (<xref ref-type="bibr" rid="B19">Liu et&#xa0;al., . 2018</xref>) added a bidirectional feature fusion module on top of FPN. Building on these approaches, BiFPN (<xref ref-type="bibr" rid="B32">Tan et&#xa0;al., 2020</xref>) introduced bidirectional connections in the process of information propagation, allowing information to flow both top-down and bottom-up within the network. This bidirectional flow effectively addresses issues of information loss and blurring in feature pyramid networks.</p>
<p>The Gold-YOLO model introduced an advanced gather and distribute mechanism (GD mechanism), which uses a unified module to collect and fuse information from all levels and distribute it to different levels, addressing the information fusion issues in traditional object detection models (<xref ref-type="bibr" rid="B38">Wang et&#xa0;al., 2024</xref>). Although multi-scale feature fusion methods have significant reference value for image processing tasks, current networks struggle to meet the practical requirements of weed detection tasks due to challenges such as limited weed features, varying lighting conditions, and occlusion problems. Therefore, developing more efficient feature fusion networks is crucial for improving the accuracy of weed detection.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Method</title>
<sec id="s3_1">
<label>3.1</label>
<title>PD-YOLO model</title>
<p>YOLOv8 is an end-to-end optimized model known for its high performance and accuracy in detection and segmentation tasks in computer vision (<xref ref-type="bibr" rid="B23">Redmon et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B24">Reis et&#xa0;al., 2023</xref>). It builds upon the improvements made in YOLOv5 (<xref ref-type="bibr" rid="B50">Zhang et&#xa0;al., 2022</xref>), and its specific structure is shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. The C2f module is a key component of YOLOv8&#x2019;s backbone network, enhancing the richness of information flow through gradient-splitting connections while maintaining a lightweight model. The neck part uses a PA-FPN structure, inspired by PANet (<xref ref-type="bibr" rid="B19">Liu et&#xa0;al., . 2018</xref>), and the head adopts a decoupled structure designed separately for object classification and bounding box regression, utilizing different loss functions to improve detection accuracy and model convergence speed. This design, combined with a dynamic sample allocation mechanism, further enhances YOLOv8&#x2019;s detection accuracy and robustness.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The overall architecture of the YOLOv8 model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g001.tif"/>
</fig>
<p>The YOLOv8 series includes multiple versions, offering more refined model parameter tuning options, making it highly effective in both high-precision and real-time applications. YOLOv8n (Nano) is the smallest and fastest version of the YOLOv8 series, suitable for mobile devices, embedded systems, and applications that require real-time processing. Its efficient performance in terms of processing speed and computational resources makes it ideal for weed detection applications.</p>
<p>Based on the lightweight YOLOv8n model, we designed an improved weed detection model&#x2014;PD-YOLO. The structure of PD-YOLO is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, and it primarily enhances the accuracy and efficiency of weed detection through the organic combination of three key components: the Backbone, the Parallel Focusing Feature Pyramid (PF-FPN), and the DyHead.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Overall architecture of FD-YOLO. PF-FPN is the feature fusion network proposed in this study, which fuses three scale features from the backbone network based on the FAFM module. Meanwhile, the HARFM module fuses low-level and high-level features within the same path and aggregates them into the FFAM module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g002.tif"/>
</fig>
<list list-type="simple">
<list-item>
<p>(1) The Backbone utilizes multiple convolutional layers to extract multi-scale features from the input image, and the C2F module enhances the feature map&#x2019;s expressive capability, effectively capturing and integrating information from different levels to support efficient and accurate object detection tasks.</p>
</list-item>
<list-item>
<p>(2) The PF-FPN is a multi-layer feature fusion pyramid that enables efficient multi-scale feature fusion, addressing the issue of similar features between weeds and plants as well as between different weeds, thereby improving the detection capability of various weed types. PF-FPN includes the Feature Filtering and Aggregation Module (FFAM) and the Hierarchical Adaptive Recalibration Fusion Module (HARFM). The FFAM module first filters and fuses features, and then extracts them, achieving multi-scale feature extraction. The HARFM module, based on attention mechanisms, further enhances feature expression, effectively improving the model&#x2019;s feature fusion capability.</p>
</list-item>
<list-item>
<p>(3) The detection head is responsible for object localization and classification based on the fused features. In field environments, weeds exhibit diverse scales and complex morphologies, which increases the difficulty of detection. To tackle these challenges, this study introduces dynamic head technology, which adaptively adjusts the parameters and structure of the detection head to more effectively capture information from different feature layers, enhancing the model&#x2019;s adaptability to complex scenarios.</p>
</list-item>
</list>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Parallel focusing feature pyramid</title>
<p>The structure of PF-FPN is shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. It consists of two parts: the FFAM module and the HARFM module. &#x201c;Fuse&#x201d; represents the process of fusing multi-scale features. The features C1, C2, and C3 are derived from different levels of the backbone network, representing information at various scales: C1 originates from lower levels, containing high resolution and rich details; C2 comes from the intermediate levels, balancing resolution and semantic information; and C3 comes from the higher levels, containing deeper semantic information despite lower resolution. The {C1, C2, C3} features are aggregated into the FFAM module, where efficient multi-scale feature fusion is achieved through attention mechanisms and convolutional layers. The fused features are then distributed across various detection scales through convolutional Downsampling or Upsampling, concatenated with features of the same level, and passed to the C2F module in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> for further fusion. The HARFM module fuses low-level and high-level features along the same path before aggregating them into the FFAM module. This parallel feature fusion, from left to right and from the center to the edges, produces the final {P1, P2, P3} feature layers, achieving complementary enhancement of multi-level features. This parallel dynamic feature fusion mechanism takes into account the diversity of feature hierarchies and effectively avoids information loss and bias through parallel fusion, significantly improving the model&#x2019;s ability to handle large-scale variations and similar feature targets. Compared to the traditional FPN, which uses a unidirectional top-down feature fusion path, this parallel fusion approach better preserves the detailed information of low-level features, avoiding the loss of detail features that may occur in traditional methods.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>The structure of the Parallel Focusing Feature Pyramid.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g003.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Feature filtering and aggregation module</title>
<p>During feature extraction, unfiltered features often introduce noise and redundant information, which can negatively impact subsequent analysis and decision-making processes. Therefore, a method of filtering before extraction is proposed. The structure of the FFAM module is shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Feature Filtering and Aggregation Module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g004.tif"/>
</fig>
<p>The filtering step for multi-scale features benefits from the ELA attention mechanism (<xref ref-type="bibr" rid="B44">Xu and Wan, 2024</xref>), which can dynamically adjust and filter based on the characteristics of the input data. It helps suppress background noise, such as soil textures, and enhances small target regions like weed leaves. Compared to SE attention (<xref ref-type="bibr" rid="B13">Hu et&#xa0;al., 2018</xref>), which only focuses on channel relationships, ELA is more suitable for agricultural scenarios with irregular spatial distributions. The feature extraction method involves using a set of parallel depthwise separable convolutions to extract multi-scale detailed features from the filtered and fused features. This channel fusion mechanism helps integrate features with different receptive field sizes, capturing extensive contextual information.</p>
<p>The input feature map F of the module consists of three features at different scales: a low-level feature <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>low</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mtext>&#xa0;R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>H</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, a high-level feature <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>high</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mtext>&#xa0;R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>H</mml:mtext>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and a mid-level feature <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>mid</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mtext>&#xa0;R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>H</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.The mathematical expression for this process is as follows, as shown in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>low</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>low</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>mid</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>D</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>mid</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>high</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>up</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>high</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes 2D convolution and; <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>D</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes downsampling using a 2D convolution with a kernel size of 3 and a stride of 2; <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>up</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes upsampling. On this basis, the high-level feature <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>high</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> generates the corresponding attention weight through the ELA attention mechanism, which is used to filter the low-level features. Meanwhile, the mid-level feature <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>mid</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> also generates corresponding attention weights through the ELA module to filter its own redundant information. Subsequently, the filtered multi-scale features are added and fused to obtain <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mtext>R</mml:mtext>
<mml:mrow>
<mml:mtext>H</mml:mtext>
<mml:mo>'</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mo>'</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The mathematical expression for this process is as follows, as shown in <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>ELA</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>high</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>low</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>high</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>ELA</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>mid</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>mid</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#xa0;"</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>ELA</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the attention weights generated by the ELA module. The ELA attention mechanism is crucial in the FFAM module, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. ELA is a novel attention mechanism that uses a simple and lightweight structure, enabling the network to precisely focus on regions of interest. It first uses adaptive average pooling to pool the input feature map x vertically and horizontally, with horizontal direction as (1, H) and vertical direction as (W, 1). For the <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msup>
<mml:mtext>c</mml:mtext>
<mml:mrow>
<mml:mtext>th</mml:mtext>
</mml:mrow>
</mml:msup>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> channel, the height h and width w are expressed as shown in <xref ref-type="disp-formula" rid="eq3">Equations 3</xref> and <xref ref-type="disp-formula" rid="eq4">4</xref>:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msubsup>
<mml:mtext>Z</mml:mtext>
<mml:mi>c</mml:mi>
<mml:mtext>h</mml:mtext>
</mml:msubsup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>h</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mtext>H</mml:mtext>
</mml:mfrac>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mtext>i</mml:mtext>
<mml:mo>&lt;</mml:mo>
<mml:mtext>w</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mtext>x</mml:mtext>
<mml:mtext>c</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>h</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>i</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msubsup>
<mml:mtext>Z</mml:mtext>
<mml:mi>c</mml:mi>
<mml:mtext>w</mml:mtext>
</mml:msubsup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>w</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mtext>w</mml:mtext>
</mml:mfrac>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mtext>j</mml:mtext>
<mml:mo>&lt;</mml:mo>
<mml:mtext>H</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mtext>x</mml:mtext>
<mml:mtext>c</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>j</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>w</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The two obtained feature vectors, <inline-formula>
<mml:math display="inline" id="im55">
<mml:mrow>
<mml:msubsup>
<mml:mtext>Z</mml:mtext>
<mml:mi>c</mml:mi>
<mml:mtext>h</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>  and <inline-formula>
<mml:math display="inline" id="im56">
<mml:mrow>
<mml:msubsup>
<mml:mtext>Z</mml:mtext>
<mml:mi>c</mml:mi>
<mml:mtext>w</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, are respectively processed through a 2D convolution, followed by a group normalization layer (GN), and finally a Sigmoid activation function to generate the attention weights. The process is illustrated in <xref ref-type="disp-formula" rid="eq5">Equations 5</xref> and <xref ref-type="disp-formula" rid="eq6">6</xref>:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>h</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>GN</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>Z</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
</mml:msubsup>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>w</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>GN</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>Z</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes 2D convolution, <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>GN</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes group normalization with 16 groups, and <inline-formula>
<mml:math display="inline" id="im14">
<mml:mtext>&#x3c3;</mml:mtext>
</mml:math>
</inline-formula> denotes the Sigmoid function.The horizontal and vertical outputs are multiplied to obtain the resulting attention weights, represented as shown in <xref ref-type="disp-formula" rid="eq7">Equation 7</xref>:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>ELA</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>h</mml:mtext>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>w</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>On the basis of initially filtered features, the module employs a set of parallel depthwise separable convolutions to extract multi-scale detailed features, enhancing the capability to capture small targets and rich semantic information, thus improving the model&#x2019;s generalization and robustness. Inspired by PKI (<xref ref-type="bibr" rid="B4">Cai et&#xa0;al., 2024</xref>), an Inception-style feature extraction module (<xref ref-type="bibr" rid="B48">Yu et&#xa0;al., 2024</xref>) is introduced, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, which demonstrates good performance in handling multi-scale target detection tasks. The large convolutional kernels can recognize larger weed shapes, while the small convolutional kernels focus on small target weeds. Compared to the single-scale convolutions in the BiFPN network, this design is more flexible in adapting to the scale variations of weed shapes. The module uses a set of parallel depthwise separable convolutions to capture small target features and contextual semantic information, with direct connections added. Specifically, according to the parameter settings of the PKI module, the optimal kernel size for the <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:msup>
<mml:mtext>m</mml:mtext>
<mml:mrow>
<mml:mtext>th</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> DWConv is set to: <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msup>
<mml:mi>k</mml:mi>
<mml:mtext>m</mml:mtext>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>m</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.The module uses a total of 4 DWConv layers, with kernel sizes of 5, 7, 9, and 11. These are followed by a 1&#xd7;1 pointwise convolution (Pwconv) to fuse the local and contextual features, generating the output feature <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mtext>R</mml:mtext>
<mml:mrow>
<mml:msup>
<mml:mtext>H</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mtext>W</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mtext>C</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The expression for <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is as follows:</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>p</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mtext>m</mml:mtext>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:msub>
<mml:mtext>D</mml:mtext>
<mml:mrow>
<mml:msup>
<mml:mtext>K</mml:mtext>
<mml:mi>m</mml:mi>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:msub>
<mml:mtext>D</mml:mtext>
<mml:mrow>
<mml:msup>
<mml:mtext>K</mml:mtext>
<mml:mi>m</mml:mi>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> the <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msup>
<mml:mtext>m</mml:mtext>
<mml:mrow>
<mml:mtext>th</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> 3&#xd7;3 depthwise separable convolution; <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>p</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the pointwise convolution. Finally, <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is combined with the original input feature <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to obtain the rich semantic output feature <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:mtext>F"</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mtext>R</mml:mtext>
<mml:mrow>
<mml:msup>
<mml:mtext>H</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mtext>W</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mtext>C</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The expression is given in <xref ref-type="disp-formula" rid="eq9">Equation 9</xref>:</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msup>
<mml:mtext>F</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Hierarchical adaptive recalibration fusion module</title>
<p>In the feature pyramid structure, high-level features contain rich semantic information due to their deep receptive fields but have lower spatial resolution, while low-level features retain high-resolution detail information but lack global semantic context. The traditional Feature Pyramid Network (FPN) fuses multi-scale features through simple linear summation, which can dilute semantic information, especially making it less sensitive to low-contrast overlapping leaf regions. Therefore, the HARFM module recalibrates the attention weights obtained by concatenating high-level and low-level features, enhancing the network&#x2019;s focus on key features and improving the overall performance of weed detection. Specifically, group normalization (GN) is used instead of batch normalization (BN) to avoid the statistical bias issues during small-batch training, which is crucial for agricultural images with complex data distributions. The HARFM structure is shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>HARFM structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g005.tif"/>
</fig>
<p>The low-level feature <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msub>
<mml:mtext>X</mml:mtext>
<mml:mtext>l</mml:mtext>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mtext>&#xa0;R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>H</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and the high-level feature <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msub>
<mml:mtext>X</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mtext>&#xa0;R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>H</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> have a large difference in the number of channels, which would result in high computational cost if concatenated directly. Therefore, the input low-level feature <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:msub>
<mml:mtext>X</mml:mtext>
<mml:mtext>l</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is passed through a 3&#xd7;3 convolution to reduce the number of channels to <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, making its dimensions consistent with <inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:msub>
<mml:mtext>X</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and then they are concatenated to obtain a feature map <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>c</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in 2 <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>H</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.The process is represented as shown in <xref ref-type="disp-formula" rid="eq10">Equation 10</xref>:</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>CBS</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>X</mml:mtext>
<mml:mtext>l</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mtext>X</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>CBS</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the low-level feature processed by 2D convolution, batch normalization (BN), and activation function (Silu). The function <inline-formula>
<mml:math display="inline" id="im33">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes concatenation. The concatenated features are then passed through a series of convolutions to generate the attention weights. To reduce computational cost, the concatenated feature is first passed through a 2D convolution to reduce the number of channels to 2 <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mtext>r</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>, resulting in <inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mtext>&#xa0;R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mtext>r</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>H</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, expressed as <xref ref-type="disp-formula" rid="eq11">Equation 11</xref>:</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3b4;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents 2D convolution and <inline-formula>
<mml:math display="inline" id="im37">
<mml:mtext>&#x3b4;</mml:mtext>
</mml:math>
</inline-formula> represents the ReLU activation function. Then, two depthwise separable convolutions are used to effectively extract low-level features while reducing the number of parameters: The process is represented as shown in <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>:</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3b4;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>D</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>&#x3b4;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>D</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im38">
<mml:mrow>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mtext>D</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents depthwise separable convolution. Finally, a 1&#xd7;1 convolution is applied to restore the channel number to <inline-formula>
<mml:math display="inline" id="im39">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, resulting in <inline-formula>
<mml:math display="inline" id="im40">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mtext>&#xa0;R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mtext>C</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>H</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The process is represented as shown in <xref ref-type="disp-formula" rid="eq13">Equation 13</xref>:</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3b4;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>d</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The ReLU activation function is introduced to improve computational efficiency. Finally, the feature map is processed through a Group Normalization (GN) layer and a Sigmoid function to generate the attention weights. The output is obtained by multiplying the original features with the weight matrix and then adding them together. The mathematical expression is given as <xref ref-type="disp-formula" rid="eq14">Equation 14</xref>:</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>out</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>GN</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Dynamic head</title>
<p>The structure of the Dynamic Head is illustrated in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>. The Dynamic Head incorporates multiple attention mechanisms: Scale-aware, Spatial-aware, and Task-aware. This design enables the Dynamic Head to address scale variations, spatial changes, and different task requirements, thereby improving the efficiency and accuracy of weed detection. Specifically, multi-level features from the feature pyramid are adjusted to the same scale and reshaped into three-dimensional tensors. The attention mechanism is applied according to the formula in <xref ref-type="disp-formula" rid="eq15">Equation 15</xref>:</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>The detailed design of the Dynamic Head and the structure of each attention module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g006.tif"/>
</fig>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mtext>W</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im41">
<mml:mtext>&#x3c0;</mml:mtext>
</mml:math>
</inline-formula> represents the attention function, implemented through fully connected layers. The attention mechanism operates along three dimensions, with each attention mechanism focusing only on a specific dimension to ensure computational efficiency. The attention function is defined as in <xref ref-type="disp-formula" rid="eq16">Equation 16</xref>:</p>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:mtext>W</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>L</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>L</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are three distinct attention functions applied to dimensions L, S, and C respectively.</p>
<p>The description of the multiple attention mechanisms&#x2014;Scale-aware, Spatial-aware, and Task-aware&#x2014;is as follows:</p>
<sec id="s3_5_1">
<label>3.5.1</label>
<title>Scale-aware attention</title>
<p>The Scale-aware Attention module is designed to handle targets of different scales by distinguishing the relative importance between feature layers and dynamically adjusting feature representations to adapt to various target scales. The input features first go through an average pooling layer to reduce the number of parameters, then are passed through a convolution layer with a kernel size of 1, using the ReLU activation function to better capture non-linear relationships in the input data. Finally, a Sigmoid activation function is applied to produce the final output. The mathematical expression is as follows in <xref ref-type="disp-formula" rid="eq17">Equation 17</xref>:</p>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>L</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>f</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mtext>SC</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext>S</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>c</mml:mtext>
</mml:mrow>
</mml:munder>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where F represents the input feature tensor, F is a linear function approximating a 1&#xd7;1 convolution, and &#x3c3; represents the Sigmoid activation function.</p>
</sec>
<sec id="s3_5_2">
<label>3.5.2</label>
<title>Spatial-aware attention</title>
<p>The Spatial-aware Attention module is designed to capture the spatial consistency of targets. This module enhances the understanding of weed locations and shapes in complex environments by identifying consistent regions across spatial positions and feature hierarchies. To reduce the dimensionality of high-dimensional features, the module operates in two steps: first, deformable convolutions are applied to achieve sparse attention learning, followed by aggregation of feature information from different levels at the same spatial locations. The mathematical expression is shown as follows in <xref ref-type="disp-formula" rid="eq18">Equation 18</xref>:</p>
<disp-formula id="eq18">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>S</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mtext>L</mml:mtext>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext>l</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mtext>L</mml:mtext>
</mml:munderover>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext>k</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mtext>K</mml:mtext>
</mml:munderover>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mrow>
<mml:mtext>l</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>k</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>l</mml:mtext>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mtext>p</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mtext>p</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>c</mml:mtext>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mtext>m</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In this context, L represents the number of feature layers, and K denotes the number of sparse sampling positions. The term <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mtext>p</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> + <inline-formula>
<mml:math display="inline" id="im46">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mtext>p</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicates the spatial offset that is self-learned to focus on a distinct region, while <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mtext>m</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> reflects the self-learned importance scalar at a specific location, <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:msub>
<mml:mtext>p</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Both <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mtext>p</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mtext>m</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> k are derived from the median-level input features of F.</p>
</sec>
<sec id="s3_5_3">
<label>3.5.3</label>
<title>Task-aware attention</title>
<p>The Task-aware attention module adapts to different detection tasks. Specifically, the input feature map x is first passed through an average pooling layer to reduce feature dimensions. Then, two fully connected layers and a normalization layer map the features into the range of -1 to 1. The normalized results are fed into a hyperfunction for further computation. This design enhances the model&#x2019;s adaptability and performance across various detection scenarios. The mathematical expression is as follows in <xref ref-type="disp-formula" rid="eq19">Equation 19</xref>:</p>
<disp-formula id="eq19">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3c0;</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mtext>max</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mtext>&#x3b2;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mtext>&#x3b2;</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the feature slice of the <inline-formula>
<mml:math display="inline" id="im52">
<mml:mrow>
<mml:msup>
<mml:mtext>c</mml:mtext>
<mml:mrow>
<mml:mtext>th</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> channel, <inline-formula>
<mml:math display="inline" id="im53">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:msup>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mtext>&#x3b2;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mtext>&#x3b2;</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">]</mml:mo>
<mml:msup>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>T</mml:mtext>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3b8;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is as a meta-function that learns to control activation thresholds through dimension reduction, neural network layers, normalization, and sigmoid transformation.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiments</title>
<sec id="s4_1">
<label>4.1</label>
<title>Datasets</title>
<p>In this study, two widely varying and challenging datasets, CottonWeedDet12 (<xref ref-type="bibr" rid="B9">Dang et&#xa0;al., 2023</xref>) and Lincoln beet (<xref ref-type="bibr" rid="B27">Salazar-Gomez et&#xa0;al., 2021</xref>), were chosen instead of a single weed dataset or datasets from specific environments. This approach comprehensively tests the detection capabilities of the PD-YOLO model under different environments and conditions, validating the generality, effectiveness, and robustness of PD-YOLO in real-world applications.</p>
<p>(1) CottonWeedDet12 is one of the largest publicly available multi-class weed detection datasets. The dataset covers 12 common weed species found in cotton fields in southern U.S. states, containing 5648 RGB images annotated for weed identification using the VGG Image Annotator (version 2.10), with a total of 9370 bounding boxes. These images were collected under natural field lighting conditions using smartphones or handheld digital cameras from June to September 2021. The dataset is characterized by weed occlusion, large scale differences, and small targets. <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> shows some original images from the CottonWeedDet12 dataset.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Illustration of images from the CottonWeedDet12 dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g007.tif"/>
</fig>
<p>(2) Lincoln beet is a dataset designed for beet and weed detection, specifically focused on addressing the challenge of occlusion. The dataset contains 4405 images with a resolution of 1902&#xd7;1080 pixels, and each image is annotated with bounding boxes for both beets and harmful weeds, totaling 39,246 bounding boxes. It is a dense dataset with small targets. These images were extracted from videos recorded in different fields in Lincoln, UK. The video recordings were conducted between May and June 2021, using two cameras, with each beet field scanned at least four times per week to capture weed development at various growth stages, showcasing different soil types, plant distributions, and weed species. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> shows some original images from the Lincoln beet dataset.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Illustration of images from the Lincoln beet dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g008.tif"/>
</fig>
<p>Although the experiments focus on cotton and beet field scenarios, the multi-scale features and weed morphology similarities of the CottonWeedDet12 dataset, along with the high density, occlusion, and small object challenges of the Lincoln Beet dataset, are highly representative and can validate the generalizability of FD-YOLO in complex agricultural environments. Additionally, the PF-FPN design concept of FD-YOLO gives it a certain level of cross-crop adaptability. The global semantic information and local detail features captured by PF-FPN can be generalized to weed detection tasks in other crops (such as corn and wheat). However, the morphological differences, planting densities, and background complexities of different crops may affect model performance, requiring further validation and optimization across datasets.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Performance experiment of PD-YOLO</title>
<p>The experiments were conducted on a 64-bit Windows 11 operating system using an NVIDIA GeForce RTX 3050Ti GPU with 16GB of memory. The PD-YOLO model was implemented in a deep learning environment using Python 3.9.16, torch 2.2.0, and CUDA 12.1. The input image size for the model was set to 640&#xd7;640, and the model was trained for 200 epochs. During training, mosaic data augmentation was applied, but it was turned off after the 15th epoch. The learning rate was set to 0.01, weight decay to 0.0005, and momentum to 0.937. The CottonWeedDet12 dataset was split in an 8:1:1 ratio for training, validation, and testing, while the Lincoln beet dataset was split in a 7:1:2 ratio.</p>
<p>We conducted experiments comparing PD-YOLO with Faster R-CNN (<xref ref-type="bibr" rid="B25">Ren et&#xa0;al., 2015</xref>), SSD (<xref ref-type="bibr" rid="B18">Liu et&#xa0;al., 2016</xref>), Yolov7-tiny (<xref ref-type="bibr" rid="B36">Wang et&#xa0;al., 2023</xref>), Yolov8n, Yolov8s, Yolov10 (<xref ref-type="bibr" rid="B37">Wang et&#xa0;al., 2024</xref>), and RT-DETR (<xref ref-type="bibr" rid="B52">Zhao et&#xa0;al., 2024</xref>) to evaluate the performance of PD-YOLO. The experiments were conducted using the CottonWeedDet12 and Lincoln beet datasets, with precision, recall, mAP@0.5, mAP@0.5:0.95, parameters, and FLOPs as evaluation metrics.</p>
<p>(1) Precision, as follows in <xref ref-type="disp-formula" rid="eq20">Equation 20</xref>, measures the accuracy of the model when predicting positive classes, which is the ratio of correctly predicted positive samples to all samples predicted as positive.</p>
<disp-formula id="eq20">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where TP (True Positive) refers to correctly identified positive samples, and FP (False Positive) refers to incorrectly identified negative samples as positive.</p>
<p>(2) Recall, as follows in <xref ref-type="disp-formula" rid="eq21">Equation 21</xref> evaluates the model&#x2019;s ability to identify positive samples, which is the ratio of correctly predicted positive samples to all actual positive samples.</p>
<disp-formula id="eq21">
<label>(21)</label>
<mml:math display="block" id="M21">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where FN (False Negative) refers to the positive samples missed by the model.</p>
<p>(3) mAP, as follows in <xref ref-type="disp-formula" rid="eq22">Equation 22</xref>, is a core evaluation metric in object detection. It calculates the Average Precision (AP) for each class, then averages them to comprehensively evaluate the model&#x2019;s detection accuracy, taking into account different classes and various IoU thresholds. The mathematical expression is as follows:</p>
<disp-formula id="eq22">
<label>(22)</label>
<mml:math display="block" id="M22">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mtext>p</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>r</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mtext>dr</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mtext>C</mml:mtext>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext>j</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mtext>c</mml:mtext>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mtext>Ap</mml:mtext>
</mml:mrow>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where c represents the number of classes, and <inline-formula>
<mml:math display="inline" id="im54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Ap</mml:mtext>
</mml:mrow>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>represents the average precision for the i class.</p>
<p>(4) FLOPs measure the hardware performance and algorithmic complexity, while FPS represents detection speed by measuring the number of frames processed per second.</p>
<p>The experimental results on the CottonWeedDet12 are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. PD-YOLO demonstrates a high precision (P) of 94.3%, outperforming all other models. Its recall (R) reached 87.0%, slightly lower than Yolov8s&#x2019;s 90.6%, but still excellent at 87.0%, showcasing its superior ability to identify targets. To evaluate detection performance under different IoU thresholds, we used the mean average precision (mAP). The results show that PD-YOLO achieved an mAP@0.5 of 95.0%, the best among all models, surpassing Faster-RCNN by 27.9%. When compared to other high-performance models such as Yolov10 and RT-DETR, PD-YOLO outperformed them by 2.3% and 2.5%, respectively. The mAP@0.5-0.95 was 88.3%, proving its ability to maintain high detection accuracy across different IoU thresholds, highlighting its strong generalization and robustness.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Performance experiment of PD-YOLO on the CottonWeedDet12 dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Precision (%)</th>
<th valign="middle" align="center">Recall (%)</th>
<th valign="middle" align="center">mAP 0.5 (%)</th>
<th valign="middle" align="center">mAP 0.5-0.95 (%)</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Faster-RCNN</td>
<td valign="middle" align="center">67.1</td>
<td valign="middle" align="center">78.6</td>
<td valign="middle" align="center">67.1</td>
<td valign="middle" align="center">63.9</td>
<td valign="top" align="center">7.05</td>
</tr>
<tr>
<td valign="middle" align="center">SSD</td>
<td valign="middle" align="center">71.9</td>
<td valign="middle" align="center">68.8</td>
<td valign="middle" align="center">71.9</td>
<td valign="middle" align="center">65.0</td>
<td valign="top" align="center">49.9</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov7-tiny</td>
<td valign="top" align="center">92.5</td>
<td valign="top" align="center">88.6</td>
<td valign="top" align="center">94.0</td>
<td valign="top" align="center">84.1</td>
<td valign="top" align="center">102.3</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov8n</td>
<td valign="middle" align="center">93.8</td>
<td valign="middle" align="center">87.6</td>
<td valign="middle" align="center">93.3</td>
<td valign="middle" align="center">86.5</td>
<td valign="middle" align="center">
<bold>109.7</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov8s</td>
<td valign="top" align="center">89.8</td>
<td valign="top" align="center">
<bold>90.6</bold>
</td>
<td valign="top" align="center">94.2</td>
<td valign="top" align="center">87.1</td>
<td valign="top" align="center">87.9</td>
</tr>
<tr>
<td valign="middle" align="center">RT-DETR</td>
<td valign="middle" align="center">90.1</td>
<td valign="middle" align="center">90.3</td>
<td valign="middle" align="center">92.5</td>
<td valign="middle" align="center">85.9</td>
<td valign="middle" align="center">31.8</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov10s</td>
<td valign="middle" align="center">89.7</td>
<td valign="middle" align="center">86.7</td>
<td valign="middle" align="center">92.7</td>
<td valign="middle" align="center">86.5</td>
<td valign="middle" align="center">72.4</td>
</tr>
<tr>
<td valign="middle" align="center">
<bold>PD-YOLO</bold>
</td>
<td valign="top" align="center">
<bold>94.3</bold>
</td>
<td valign="top" align="center">87.0</td>
<td valign="top" align="center">
<bold>95.0</bold>
</td>
<td valign="top" align="center">
<bold>88.3</bold>
</td>
<td valign="middle" align="center">42.5</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values in the table indicate the optimal performance of each method</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The experimental results on the Lincoln beet dataset are shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. Compared with the baseline model Yolov8n, PD-YOLO achieved improvements of 0.6% in precision (P), 1% in recall (R), 1.3% in mAP@0.5, and 0.9% in mAP@0.5:0.95, demonstrating that the optimizations in PD-YOLO effectively enhance model performance. PD-YOLO outperforms most comparison models in both precision and recall. Specifically, PD-YOLO achieved a precision of 75.4%, 3.9% higher than Faster-RCNN, 13.1% higher than SSD, and slightly higher than Yolov7-tiny. In terms of recall, PD-YOLO reached 71.4%, equal to Yolov10s and 3.6% higher than Faster-RCNN.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance experiment of PD-YOLO on the Lincoln beet dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Precision (%)</th>
<th valign="middle" align="center">Recall (%)</th>
<th valign="middle" align="center">mAP 0.5 (%)</th>
<th valign="middle" align="center">mAP 0.5-0.95 (%)</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Faster-RCNN</td>
<td valign="middle" align="center">71.5</td>
<td valign="middle" align="center">67.8</td>
<td valign="middle" align="center">71.4</td>
<td valign="middle" align="center">49.8</td>
<td valign="top" align="center">6.90</td>
</tr>
<tr>
<td valign="middle" align="center">SSD</td>
<td valign="middle" align="center">62.3</td>
<td valign="middle" align="center">63.0</td>
<td valign="middle" align="center">62.3</td>
<td valign="middle" align="center">39.5</td>
<td valign="top" align="center">49.9</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov7-tiny</td>
<td valign="top" align="center">76.5</td>
<td valign="top" align="center">71.0</td>
<td valign="top" align="center">76.4</td>
<td valign="top" align="center">51.0</td>
<td valign="top" align="center">
<bold>108.9</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov8n</td>
<td valign="middle" align="center">74.8</td>
<td valign="middle" align="center">70.4</td>
<td valign="middle" align="center">75.5</td>
<td valign="middle" align="center">52.7</td>
<td valign="middle" align="center">101.8</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov8s</td>
<td valign="top" align="center">75.6</td>
<td valign="top" align="center">
<bold>71.9</bold>
</td>
<td valign="top" align="center">
<bold>76.9</bold>
</td>
<td valign="top" align="center">
<bold>53.7</bold>
</td>
<td valign="top" align="center">86.6</td>
</tr>
<tr>
<td valign="middle" align="center">RT-DETR</td>
<td valign="middle" align="center">
<bold>77.8</bold>
</td>
<td valign="middle" align="center">70.3</td>
<td valign="middle" align="center">73.6</td>
<td valign="middle" align="center">51.4</td>
<td valign="middle" align="center">29.9</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov10s</td>
<td valign="middle" align="center">74.3</td>
<td valign="middle" align="center">71.4</td>
<td valign="middle" align="center">75.5</td>
<td valign="middle" align="center">51.9</td>
<td valign="middle" align="center">67.3</td>
</tr>
<tr>
<td valign="middle" align="center">PD-YOLO</td>
<td valign="middle" align="center">75.4</td>
<td valign="middle" align="center">71.4</td>
<td valign="middle" align="center">76.8</td>
<td valign="middle" align="center">53.6</td>
<td valign="middle" align="center">42.9</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values in the table indicate the optimal performance of each method</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The mAP@0.5 reached 76.8%, slightly lower than Yolov8s&#x2019;s 76.9%, but still 3.2% and 1.2% higher than RT-DETR and Yolov10s, respectively. Regarding mAP@0.5:0.95,PD-YOLO also led with a score of 53.6%, outperforming SSD, RT-DETR, and YOLOv10 by 14.1%, 2.2%, and 1.7%, respectively. These results demonstrate that PD-YOLO maintains high detection accuracy even under more stringent evaluation criteria, showcasing the model&#x2019;s robustness and broad adaptability.</p>
<p>FPS measures the number of image frames processed by the model per second, which is a critical metric for evaluating real-time performance. PD-YOLO achieved FPS values of 42.5 and 42.9 on the CottonWeedDet12 and Lincoln beet test sets, respectively, meeting the requirements for real-time performance. <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> presents the parameter counts and computational complexity of different detection models. Compared to some lightweight models, such as YOLOv7-tiny, which achieved FPS values of 102.3 and 108.9 on the CottonWeedDet12 and Lincoln beet test sets, respectively, with a computational complexity of 13.1 GFLOPs and 6.04M parameters, PD-YOLO has a lower FPS. However, with a computational complexity of 10.6 GFLOPs and 3.96M parameters, PD-YOLO maintains a moderate level of computational complexity and parameter count, achieving a good balance between real-time performance and resource requirements.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Parameter count and computational complexity of different detection models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Parameters(M)</th>
<th valign="middle" align="center">GFLOPS(G)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Faster-RCNN</td>
<td valign="middle" align="center">41.41</td>
<td valign="middle" align="center">121.4</td>
</tr>
<tr>
<td valign="middle" align="center">SSD</td>
<td valign="middle" align="center">14.50</td>
<td valign="middle" align="center">15.8</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov7-tiny</td>
<td valign="top" align="center">6.04</td>
<td valign="top" align="center">13.1</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov8n</td>
<td valign="middle" align="center">3.01</td>
<td valign="middle" align="center">8.1</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov8s</td>
<td valign="top" align="center">11.20</td>
<td valign="top" align="center">28.5</td>
</tr>
<tr>
<td valign="middle" align="center">RT-DETR</td>
<td valign="middle" align="center">19.89</td>
<td valign="middle" align="center">57.0</td>
</tr>
<tr>
<td valign="middle" align="center">Yolov10s</td>
<td valign="middle" align="center">8.04</td>
<td valign="middle" align="center">24.5</td>
</tr>
<tr>
<td valign="middle" align="center">PD-YOLO</td>
<td valign="middle" align="center">3.96</td>
<td valign="middle" align="center">10.6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In summary, the PD-YOLO model outperformed other detection models on most metrics in the CottonWeedDet12 and Lincoln Beet datasets, especially in terms of precision and mAP@0.5:0.95. The model provides efficient processing speeds while maintaining moderate computational demands, making it suitable for resource-constrained environments. It achieves an excellent balance between detection accuracy and real-time performance.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Ablation study</title>
<p>We introduced the TIDE metric to comprehensively evaluate the impact and performance of different components on the PD-YOLO model. The TIDE metric allows us to gain a more in-depth and holistic understanding of the role each component plays within the overall detection system, as well as the interpretability of the model design. The effectiveness of weed detection is significantly influenced by the size and quality of the dataset used. Compared to the Lincoln beet dataset, which only annotates two classes (weeds and plants), the CottonWeedDet12 dataset provides detailed annotations of 12 different weed categories, making it more challenging. Given this, we selected the CottonWeedDet12 dataset for the ablation study.</p>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Comparison of different multi-scale feature fusion strategies</title>
<p>Considering the morphological similarity of weeds, we designed the Parallel Focusing Feature Pyramid (PF-FPN). To demonstrate the ability of PF-FPN in multi-scale feature fusion, we conducted comparative experiments with FPN (<xref ref-type="bibr" rid="B17">Lin et&#xa0;al., 2017</xref>), PA-FPN (<xref ref-type="bibr" rid="B19">Liu et&#xa0;al., . 2018</xref>), BiFPN (<xref ref-type="bibr" rid="B32">Tan et&#xa0;al., 2020</xref>), and AFPN (<xref ref-type="bibr" rid="B45">Yang et&#xa0;al., 2023</xref>). The experimental results are shown in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Performance results of different feature fusion methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Parameters (M)</th>
<th valign="middle" align="center">GFLOPS (G)</th>
<th valign="middle" align="center">Precision (%)</th>
<th valign="middle" align="center">Recall (%)</th>
<th valign="middle" align="center">mAP0.5 (%)</th>
<th valign="middle" align="center">mAP0.5-0.95 (%)</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">FPN</td>
<td valign="middle" align="center">3.15</td>
<td valign="middle" align="center">9.0</td>
<td valign="middle" align="center">93.4</td>
<td valign="middle" align="center">86.1</td>
<td valign="middle" align="center">93.2</td>
<td valign="top" align="center">86.8</td>
<td valign="top" align="center">56.4</td>
</tr>
<tr>
<td valign="top" align="center">BIFPN</td>
<td valign="top" align="center">2.00</td>
<td valign="top" align="center">7.1</td>
<td valign="top" align="center">92.5</td>
<td valign="top" align="center">87.4</td>
<td valign="top" align="center">93.2</td>
<td valign="top" align="center">85.9</td>
<td valign="top" align="center">97.6</td>
</tr>
<tr>
<td valign="top" align="center">PA-FPN</td>
<td valign="top" align="center">3.49</td>
<td valign="top" align="center">9.6</td>
<td valign="top" align="center">92.4</td>
<td valign="top" align="center">88.2</td>
<td valign="top" align="center">94.2</td>
<td valign="top" align="center">87.7</td>
<td valign="top" align="center">56.3</td>
</tr>
<tr>
<td valign="top" align="center">AFPN</td>
<td valign="top" align="center">2.60</td>
<td valign="top" align="center">8.4</td>
<td valign="top" align="center">93.7</td>
<td valign="top" align="center">87.2</td>
<td valign="top" align="center">93.1</td>
<td valign="top" align="center">86.8</td>
<td valign="top" align="center">54.4</td>
</tr>
<tr>
<td valign="top" align="center">PF-FPN</td>
<td valign="middle" align="center">3.96</td>
<td valign="middle" align="center">10.6</td>
<td valign="top" align="center">94.3</td>
<td valign="top" align="center">87.0</td>
<td valign="top" align="center">95.0</td>
<td valign="top" align="center">88.3</td>
<td valign="middle" align="center">42.5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The experimental results in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> show that PF-FPN demonstrates significant advantages across multiple metrics compared to other multi-scale feature fusion methods. Specifically, PF-FPN&#x2019;s mAP@0.5 is 1.8% higher than FPN, 0.7% higher than BiFPN, 1.9% higher than AFPN, and 0.8% higher than PA-FPN. In terms of mAP@0.5:0.95, PF-FPN also achieves the highest score, reaching 88.3%. Compared to other methods, PF-FPN&#x2019;s mAP@0.5:0.95 is 1.5% higher than FPN and AFPN, 2.4% higher than BiFPN, and 0.6% higher than PA-FPN.</p>
<p>Although the parameter count and computational complexity are higher, resulting in a lower FPS compared to other methods, the significant improvements in precision and recall demonstrate PF-FPN&#x2019;s superiority in multi-scale feature fusion. These results suggest that PF-FPN can more effectively fuse multi-scale features, leading to a substantial improvement in the model&#x2019;s detection performance.</p>
<p>We used the Grad-CAM (<xref ref-type="bibr" rid="B28">Selvaraju et&#xa0;al., 2017</xref>) heatmap visualization method to present the results in the form of heatmaps, which helps improve the interpretability and reliability of the network. <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> shows the heatmaps generated by different multi-scale methods. The first row (a-e) represents the original images, while the second to fifth rows show the heatmaps generated by different models. In the heatmaps, darker regions indicate where the model&#x2019;s attention is more focused. Compared to other methods, the heatmap generated by PD-YOLO shows a more pronounced focus on weed regions. This concentrated attention helps better capture multi-scale features in the image, thereby improving detection accuracy, particularly when dealing with small objects and reducing the likelihood of missed detections.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Heatmaps of different images. The first row <bold>(a-d)</bold> shows the original images of different weeds, while the second to fifth rows <bold>(a-d)</bold> display the corresponding heatmaps generated by the model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g009.tif"/>
</fig>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Comparison of different modules</title>
<p>This section presents the detailed experimental results of the proposed PD-YOLO method. We conducted a comprehensive comparison of PD-YOLO, including the FFAM module, HARFM module, and Dyhead framework, with the baseline model YOLOv8n. We evaluated the performance differences of the baseline model when using and not using FFAM, HARFM, and Dyhead. To investigate the specific impact of each module on model performance, we treated FFAM and Dyhead as independent functional modules based on YOLOv8n, with the FFAM and HARFM modules together forming PF-FPN. Subsequently, by applying the controlled variable method, we analyzed the performance improvements of these modules on the CottonWeedDet12 dataset.</p>
<p>By introducing the TIDE metric (<xref ref-type="bibr" rid="B3">Bolya et&#xa0;al., 2020</xref>) for model evaluation and conducting a series of carefully designed ablation experiments, we validated the functionality and performance of each module in PD-YOLO. The TIDE evaluation method identifies the following types of errors in single-class detection problems:</p>
<list list-type="simple">
<list-item>
<p>(1) Classification Error (Cls): The model correctly locates the object but misclassifies its category.</p>
</list-item>
<list-item>
<p>(2) Localization Error (Loc): The model correctly identifies the target category, but the bounding box is inaccurate.</p>
</list-item>
<list-item>
<p>(3) Classification and Localization Error (Both): The model makes errors in both classification and localization.</p>
</list-item>
<list-item>
<p>(4) Duplicate Detection Error (Dupl): The model generates multiple high-scoring bounding boxes for the same object.</p>
</list-item>
<list-item>
<p>(5) Background Misclassification (Bkg): The model mistakenly classifies a generated bounding box as the background.</p>
</list-item>
<list-item>
<p>(6) Missed Ground Truth Bounding Box (Miss): The model fails to detect an object that actually exists.</p>
</list-item>
<list-item>
<p>(7) False Positive (FP): The model incorrectly classifies a negative instance as a positive one.</p>
</list-item>
<list-item>
<p>(8) False Negative (FN): The model incorrectly classifies a positive instance as a negative one.</p>
</list-item>
</list>
<p>
<xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref> presents the results of ablation experiments conducted on the CottonWeedDet12 dataset, showing significant improvements in several key metrics compared to the original YOLOv8n baseline. Specifically, to address the issue of missed and false detections caused by morphological similarity, the FAFM module was introduced, resulting in increases of 0.5% in mAP@0.5 and 0.9% in mAP@0.5:0.95, demonstrating the effectiveness of FAFM in enhancing small object detection. The HARFM module strengthens weed feature representations, improving the model&#x2019;s accuracy. The combination of the HARFM and FAFM modules forms PF-FPN, which shows significant improvements in both mAP@0.5 and mAP@0.5:0.95, reaching 94.2% and 87.4%, respectively. This indicates that PF-PFN enhances the performance of multi-scale feature fusion in weed detection. The Dyhead architecture was introduced to improve the model&#x2019;s stability and accuracy.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Performance results of the ablation experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Yolov8</th>
<th valign="middle" align="center">FAFM</th>
<th valign="middle" align="center">HARFM</th>
<th valign="middle" align="center">Dyhead</th>
<th valign="middle" align="center">Parameters (M)</th>
<th valign="middle" align="center">GFLOPS (G)</th>
<th valign="middle" align="center">P (%)</th>
<th valign="middle" align="center">R (%)</th>
<th valign="middle" align="center">map0.5 (%)</th>
<th valign="middle" align="center">map0.5-0.95 (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">3.01</td>
<td valign="middle" align="center">8.1</td>
<td valign="middle" align="center">93.8</td>
<td valign="middle" align="center">87.6</td>
<td valign="middle" align="center">93.3</td>
<td valign="middle" align="center">86.5</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">3.10</td>
<td valign="middle" align="center">8.4</td>
<td valign="middle" align="center">93.1</td>
<td valign="middle" align="center">89.3</td>
<td valign="middle" align="center">93.8</td>
<td valign="middle" align="center">87.6</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">3.49</td>
<td valign="middle" align="center">9.6</td>
<td valign="middle" align="center">92.4</td>
<td valign="middle" align="center">88.2</td>
<td valign="middle" align="center">94.2</td>
<td valign="middle" align="center">87.7</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">3.51</td>
<td valign="middle" align="center">9.3</td>
<td valign="middle" align="center">93.4</td>
<td valign="middle" align="center">87.7</td>
<td valign="middle" align="center">94.2</td>
<td valign="middle" align="center">87.4</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">3.96</td>
<td valign="middle" align="center">10.6</td>
<td valign="middle" align="center">94.3</td>
<td valign="middle" align="center">87.0</td>
<td valign="middle" align="center">95.0</td>
<td valign="middle" align="center">88.3</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>"&#x2713;" indicates the activation of the corresponding module.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Compared to the baseline model, PD-YOLO showed a 0.6% decrease in recall (R), while precision (P), mAP@0.5, and mAP@0.5:0.95 increased by 0.5%, 1.7%, and 1.8%, respectively. PD-YOLO&#x2019;s parameter count increased by 0.95M, and GFLOPs increased by 2.5G, resulting in only marginal computational cost increases but significantly better performance across multiple key metrics.</p>
<p>
<xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref> presents the results evaluated using the TIDE method, providing a deeper understanding of the performance improvements in the modified model. The YOLOv8n model had a relatively high background misclassification rate (Bkg). After adding the FAFM module, the Bkg rate decreased by 0.6%. As shown in the TIDE metric statistics in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, the proportion of background misclassification significantly decreased, demonstrating that FAFM effectively reduces background errors in weed detection.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>TIDE metrics of the ablation experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Yolov8</th>
<th valign="middle" align="center">FAFM</th>
<th valign="middle" align="center">HARFM</th>
<th valign="middle" align="center">Dyhead</th>
<th valign="middle" align="center">Cls (%)</th>
<th valign="middle" align="center">Loc (%)</th>
<th valign="middle" align="center">Both (%)</th>
<th valign="middle" align="center">Dupl (%)</th>
<th valign="middle" align="center">Bkg (%)</th>
<th valign="middle" align="center">Miss (%)</th>
<th valign="middle" align="center">FP (%)</th>
<th valign="middle" align="center">FN (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">1.30</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.06</td>
<td valign="middle" align="center">0.11</td>
<td valign="middle" align="center">1.47</td>
<td valign="middle" align="center">1.30</td>
<td valign="middle" align="center">4.74</td>
<td valign="middle" align="center">2.06</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">1.05</td>
<td valign="middle" align="center">1.87</td>
<td valign="middle" align="center">0.01</td>
<td valign="middle" align="center">0.06</td>
<td valign="middle" align="center">0.97</td>
<td valign="middle" align="center">1.66</td>
<td valign="middle" align="center">2.73</td>
<td valign="middle" align="center">3.76</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">1.18</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.04</td>
<td valign="middle" align="center">1.18</td>
<td valign="middle" align="center">1.25</td>
<td valign="middle" align="center">4.03</td>
<td valign="middle" align="center">2.00</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">0.87</td>
<td valign="middle" align="center">0.99</td>
<td valign="middle" align="center">0.09</td>
<td valign="middle" align="center">0.11</td>
<td valign="middle" align="center">1.47</td>
<td valign="middle" align="center">1.11</td>
<td valign="middle" align="center">3.76</td>
<td valign="middle" align="center">2.06</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">0.87</td>
<td valign="middle" align="center">0.69</td>
<td valign="middle" align="center">0.02</td>
<td valign="middle" align="center">0.06</td>
<td valign="middle" align="center">1.47</td>
<td valign="middle" align="center">1.10</td>
<td valign="middle" align="center">3.31</td>
<td valign="middle" align="center">1.75</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>"&#x2713;" indicates the activation of the corresponding module.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Statistical charts of various TIDE metrics. <bold>(a)</bold> Yolov8, <bold>(b)</bold> Yolov8n+FAFM, <bold>(c)</bold> Yolov8n+Dyhead, <bold>(d)</bold> Yolov8+FAFM+HARFM, <bold>(e)</bold> Yolov8+FAFM+HARFM+Dyhead.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g010.tif"/>
</fig>
<p>After combining FAFM and HARFM, the model performed well across multiple metrics, achieving low classification error (Cls) and missed detection (Miss) rates of 0.87% and 0.99% respectively. This indicates that PF-FPN classifies similar weeds more accurately.</p>
<p>The PD-YOLO model excels in multiple aspects, with reductions in Cls, localization error (Loc), both classification and localization errors (Both), duplicate detection (Dupl), missed detections (Miss), false positives (FP), and false negatives (FN) by 0.43%, 0.17%, 0.04%, 0.05%, 0.2%, 1.43%, and 0.31%, respectively, while Bkg remained unchanged. This shows that PF-FPN enhances the accuracy of classifying similar weeds. As shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, PD-YOLO demonstrates superior performance in both localization and classification, with only 0.02% of Dupl errors, indicating that the model rarely makes simultaneous errors in classification and localization. This further validates the improvements in the model&#x2019;s accuracy and its ability to reduce the risks of misclassification, localization errors, and missed detections.</p>
<p>The PD-YOLO model significantly reduces false positive and background error rates while maintaining high precision and recall, thus improving overall detection performance, though its processing speed is slower. These experimental results clearly demonstrate that, compared to the original YOLOv8n algorithm, the PD-YOLO model significantly optimizes and enhances performance, validating the effectiveness of the algorithm improvements proposed in this study.</p>
<p>We evaluated the performance of weed detection in various scenarios, including dense weed clusters, partial occlusion, multi-class detection, small-sized weeds, and other complex conditions. The detection results were visualized and compared to observe the algorithm&#x2019;s ability to identify targets in terms of location, size, and category information. <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> shows some of the results, where the first row displays the original YOLOv8 results and the second row shows the improved PD-YOLO model&#x2019;s detection outcomes. In <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>, due to the small weed target on the left edge, YOLOv8 exhibited missed detections and false positives, while the improved model successfully avoided these issues. In <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>, the model was able to address detection errors caused by morphological differences within the same weed species, resulting in more accurate bounding boxes. In <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>, the improved model reduced false positives in scenarios involving occlusion. <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> presents a challenging sample with dense, small targets, and the improved model significantly enhanced detection performance in this difficult scenario.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>
<bold>(a-d)</bold> represent different weed images. The first row shows the YOLOv8 results, and the second row shows the FD-YOLO results. Green boxes represent correct detections, blue boxes represent false detections, and red boxes represent missed detections. Compared to YOLOv8, FD-YOLO reduces missed detections in small target edge weeds <bold>(a)</bold>, morphological differences <bold>(b)</bold>, Mutually occluded weeds <bold>(c)</bold>, and dense occlusion <bold>(d)</bold> scenarios, with more accurate bounding boxes, thanks to the multi-scale feature fusion of PF-FPN and the dynamic attention mechanism of DyHead.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1506524-g011.tif"/>
</fig>
<p>With these improvements, the model&#x2019;s detection performance in complex conditions was significantly enhanced. In practical applications, especially when detecting multiple weed species in complex environments, PD-YOLO can more reliably identify and locate targets, reducing both false positives and missed detections.</p>
</sec>
</sec>
</sec>
<sec id="s5" sec-type="discussion">
<label>5</label>
<title>Discussion</title>
<p>This study improves and optimizes the YOLOv8 model, enhancing its performance in weed detection tasks and developing a novel weed detection model, PD-YOLO. The FAFM and HARFM modules were introduced in this study, and based on these modules, a Parallel Focusing Feature Pyramid was proposed to replace the original PA-FPN, improving the model&#x2019;s feature fusion capability. Additionally, a dynamic detection head was incorporated to enhance the model&#x2019;s detection stability. These methods improve the classification accuracy of the traditional YOLOv8n model in weed detection and reduce both missed detections and false positives.</p>
<p>Section 4.1 focuses on the performance comparison of different object detection models. The results in <xref ref-type="table" rid="T1">
<bold>Tables&#xa0;1</bold>
</xref> and <xref ref-type="table" rid="T2">
<bold>2</bold>
</xref> show that FD-YOLO achieves an mAP@0.5 of 95.0% on the CottonWeedDet12 dataset and 76.8% on the Lincoln Beet dataset. This difference may be attributed to several factors: first, the Lincoln Beet dataset only contains two labeled categories&#x2014;&#x201d;weeds&#x201d; and &#x201c;beetroot&#x201d;&#x2014;while CottonWeedDet12 includes 12 subcategories of weeds, which tests the model&#x2019;s fine-grained classification ability; second, the weed distribution in the Lincoln Beet dataset is denser, and the target sizes are smaller, leading to increased difficulty in localization in dense occlusion scenarios; furthermore, the images in the Lincoln Beet dataset were captured under the variable field lighting conditions in the UK, with diverse background soil types, further increasing the detection complexity. FD-YOLO&#x2019;s PF-FPN enhances multi-category feature discrimination through the FFAM and HARFM modules, demonstrating clear advantages on CottonWeedDet12, but the high-density targets in Lincoln Beet require stronger spatial context modeling capabilities. The current model still has room for improvement in the spatial awareness attention mechanism of the Dynamic Head (DyHead). Future work could focus on introducing adaptive resolution adjustment strategies or enhancing spatial attention weight distribution to further improve the model&#x2019;s performance in high-density scenarios. These differences indicate that FD-YOLO requires fine-tuning or data augmentation for specific environments in cross-regional and multi-crop scenarios.</p>
<p>The results in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> show that PD-YOLO has fewer parameters and lower computational complexity than lightweight models like YOLOv7-tiny, positioning it as an efficient lightweight model. However, its FPS performance still lags significantly, and it may face challenges such as insufficient frame rates and image blur in high-speed agricultural robots. Further optimization of computational efficiency or the adoption of hardware acceleration solutions is needed. Additionally, future research could explore techniques such as model pruning, quantization, and hardware acceleration to better adapt to low-power embedded devices, ensuring its wide applicability in real-time agricultural applications.</p>
<p>In practical agricultural robotics applications, FD-YOLO can integrate with SLAM technology to enable &#x201c;detection-navigation-operation&#x201d; integration. Following Zhang W et&#xa0;al (<xref ref-type="bibr" rid="B49">Zhang et&#xa0;al., 2024</xref>), combining 2D LiDAR and visual sensors with YOLOv3 algorithm allows target detection and information mapping onto 2D grid maps for efficient path planning, eliminating computational latency-induced trajectory deviations in traditional models. Researchers applied the trained DIN-LW-YOLO model to autonomous laser weeding robots in strawberry fields, with robot speed set at 0.50 m/s and Intel Realsense D435i camera mounted 600 mm above ground at 30 fps. Field tests demonstrated 92.6% weed control rate and 1.2% seedling damage rate (<xref ref-type="bibr" rid="B51">Zhao et&#xa0;al., 2025</xref>). Moreover, FD-YOLO&#x2019;s lightweight design enables edge device deployment for future weed management. Similar studies deployed customized YOLOv7 models on NVIDIA Jetson Xavier NX platforms, integrating robotic frame spraying systems that recognize Amaranthus palmeri in cornfields for real-time spot spraying (<xref ref-type="bibr" rid="B1">Balabantaray et&#xa0;al., 2024</xref>).</p>
<p>Furthermore, the ablation experiments validated the roles and necessity of each module and their impact on model size, as discussed in Section 4.2. The experimental results demonstrated the impact of each module on the model&#x2019;s recognition accuracy, as well as a comparison of different feature pyramids. In PF-FPN, the excessive upsampling and downsampling during feature aggregation and distribution by the FFAM module led to a loss of detailed features. Therefore, the introduction of the HARFM module reduced this loss, showcasing efficient feature fusion capabilities. Despite the increase in computational cost and detection speed, the improvements in multi-scale feature fusion and small object detection enable the PD-YOLO model to perform well in scenarios involving high-density, partial occlusion, and multi-class weeds. However, in practical applications, environmental adaptability still needs to be considered. The robustness of the model under low-light conditions or extreme weather has not been validated. The current dataset is primarily based on natural lighting conditions, which may limit the model&#x2019;s stability in complex lighting scenarios.</p>
<p>The systematic analysis of the TIDE metrics provides clear directions for model optimization. Experimental results show that FD-YOLO significantly outperforms the baseline model in terms of classification errors and localization errors, but there is still room for further optimization. To address classification errors, future work could introduce fine-grained feature alignment strategies to enhance the model&#x2019;s ability to distinguish between morphologically similar weeds. The residual localization errors may stem from blurred object boundaries in complex occlusion scenarios, which could be improved by integrating deformable convolutions or refining the bounding box regression loss function to boost localization accuracy. Additionally, the background misdetection rate remains relatively high, indicating that the current model lacks sufficient suppression of background noise such as soil texture. To address this, more diverse background samples should be included during data augmentation, or a lightweight background-aware attention module could be designed.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusions and future work</title>
<p>This study proposes PD-YOLO, a novel computer vision method specifically designed for real-time weed detection. The architecture of PD-YOLO combines the Parallel Focusing Feature Pyramid (PF-FPN) and Dyhead framework, built upon the YOLOv8n framework. The FAFM module optimizes feature fusion, enhancing the model&#x2019;s representational capabilities, while the HARFM module strengthens weed-specific features, improving weed identification. The PF-FPN network, developed with consideration of weed morphological characteristics, serves as an effective feature fusion network for weed detection. The Dyhead framework improves the design of the detection head, ensuring both accuracy and stability in detection results.</p>
<p>The research results show that, compared to the baseline model, PD-YOLO improves mAP by 1.7% and 1.8% (at thresholds of 0.5 and 0.5-0.95, respectively). While maintaining a lightweight structure, PD-YOLO outperforms current mainstream object detection algorithms, demonstrating superior performance. Moreover, although the model&#x2019;s detection speed meets real-time detection requirements, there is potential for further optimization in real-world field environments.</p>
<p>Future research will focus on the following directions:</p>
<p>(1)Data Augmentation and Multimodal Fusion: Integrating multispectral imaging data to enhance the model&#x2019;s detection capability under complex lighting and occlusion conditions, and expanding data diversity through synthetic data augmentation (e.g., simulating rain, fog, and shadows).</p>
<p>Lightweight and Efficiency Optimization: Developing an FD-YOLO-Tiny variant that combines the vMamba (<xref ref-type="bibr" rid="B54">Zhu et&#xa0;al., 2024</xref>) architecture to improve the model&#x2019;s backbone, reducing computational overhead while maintaining accuracy, and adapting it for deployment on edge devices.</p>
<p>(2)Error-Driven Model Improvement: Based on the analysis results from the TIDE metrics, specifically optimizing the false negative and false positive modules. This can be achieved by strengthening the training of negative samples, improving the loss function, or adjusting the post-processing stage of the detection algorithm to reduce misdetections and missed detections.</p>
<p>(3)Cross-Scene Validation and Transfer Learning: Expanding experimental validation to include different crops, such as corn and wheat, and varying agricultural environments. Combining transfer learning techniques to enhance the model&#x2019;s generalization ability, ensuring its practicality in diverse agricultural settings.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: (1) CottonWeedDet12, Dang F, Chen D, Lu Y, et&#xa0;al. YOLOWeeds: A novel benchmark of YOLO object detectors for multi-class weed detection in cotton production systems. Computers and Electronics in Agriculture, 2023, 205: 107655.(2) Lincoln beet, Salazar-Gomez A, Darbyshire M, Gao J, et&#xa0;al. Towards practical object detection for weed spraying in precision agriculture. arxiv preprint arxiv:2109.11048, 2021.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>SL: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. ZC: Conceptualization, Investigation, Validation, Visualization, Writing &#x2013; review &amp; editing. JX: Data curation, Investigation, Validation, Visualization, Writing &#x2013; review &amp; editing. HZ: Investigation, Validation, Visualization, Writing &#x2013; review &amp; editing. JG: Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Validation.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work is supported in part by the Dongguan Science and Technology of Social Development Program (20221800905102), Project of Education Department of Guangdong Province (2022ZDZX4053), and National College Student Innovation and Entrepreneurship Training Program (202311819022).</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>During the preparation of this work the authors used ChatGPT in order to improve language and readability. After using this tool/service, the authors reviewed and edited the content as needed and take(s) full responsibility for the content of the publication.</p>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balabantaray</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Behera</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Liew</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chamara</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Jhala</surname> <given-names>A. J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Targeted weed management of Palmer amaranth using robotics and deep learning (YOLOv7)</article-title>. <source>Front. Robotics AI</source> <volume>11</volume>, <elocation-id>1441371</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/frobt.2024.1441371</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bansal</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Olsen</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Estev&#xe3;o</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Remote sensing for weed detection and control</article-title>. <source>arXiv preprint arXiv</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2410.22554</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bolya</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Foley</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hays</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hoffman</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Tide: A general toolbox for identifying object detection errors</article-title>,&#x201d; in <conf-name>Computer Vision&#x2013;ECCV 2020: 16th European Conference, Glasgow, UK</conf-name>, <italic>Proceedings, Part III 16</italic>. (<publisher-loc>Glasgow, UK</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>558</fpage>&#x2013;<lpage>573</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Poly kernel inception network for remote sensing detection</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source> (<publisher-loc>Seattle,WA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>27706</fpage>&#x2013;<lpage>27716</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>You only look one-level feature</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name> (<publisher-loc>Nashville, TN</publisher-loc>, <publisher-name>IEEE Computer Society</publisher-name>), <fpage>13039</fpage>&#x2013;<lpage>13048</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lian</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Weed biology and management in the multi-omics era: progress and perspectives</article-title>. <source>Plant Commun</source>. <volume>5</volume> (<issue>4</issue>). doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.xplc.2024.100816</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Accurate leukocyte detection based on deformable-DETR and multi-level feature fusion for aiding diagnosis of blood diseases</article-title>. <source>Comput. Biol. Med.</source> <volume>170</volume>, <fpage>107917</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.107917</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). &#x201c;<article-title>Dynamic head: Unifying object detection heads with attentions</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name> (<publisher-loc>Nashville, TN</publisher-loc>, <publisher-name>IEEE Computer Society</publisher-name>), <fpage>7373</fpage>&#x2013;<lpage>7382</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>YOLOWeeds: A novel benchmark of YOLO object detectors for multi-class weed detection in cotton production systems</article-title>. <source>Comput. Electron. Agric.</source> <volume>205</volume>, <fpage>107655</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.107655</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deng</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Weed database development: An updated survey of public weed datasets and cross-season weed detection adaptation</article-title>. <source>Ecol. Inf.</source> <volume>81</volume>, <fpage>102546</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2024.102546</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Identifying rice field weeds from unmanned aerial vehicle remote sensing imagery using deep learning</article-title>. <source>Plant Methods</source> <volume>20</volume>, <fpage>105</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-024-01232-0</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hasan</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Sohel</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Diepeveen</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Laga</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>M. G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A survey of deep learning techniques for weed detection from images</article-title>. <source>Comput. Electron. Agric.</source> <volume>184</volume>, <fpage>106067</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2021.106067</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Squeeze-and-excitation networks</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>7132</fpage>&#x2013;<lpage>7141</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Coleman</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Bender</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Deep learning techniques for in-crop weed recognition in large-scale grain production systems: a review</article-title>. <source>Precis. Agric.</source> <volume>25</volume>, <fpage>1</fpage>&#x2013;<lpage>29</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11119-023-10073-1</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Islam</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Rashid</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Wibowo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C. Y.</given-names>
</name>
<name>
<surname>Morshed</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wasimi</surname> <given-names>S. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Early weed detection using image processing and machine learning techniques in an Australian chilli farm</article-title>. <source>Agriculture</source> <volume>11</volume>, <fpage>387</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture11050387</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kudsk</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Streibig</surname> <given-names>J. C.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Herbicides&#x2013;a two-edged sword</article-title>. <source>Weed Res.</source> <volume>43</volume>, <fpage>90</fpage>&#x2013;<lpage>102</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1046/j.1365-3180.2003.00328.x</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>T. Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Hariharan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Belongie</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Feature pyramid networks for object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>2117</fpage>&#x2013;<lpage>2125</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Anguelov</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Erhan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Reed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>C. Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). &#x201c;<article-title>Ssd: Single shot multibox detector</article-title>,&#x201d; in <conf-name>Computer Vision&#x2013;ECCV 2016: 14th European Conference</conf-name>, <conf-loc>Amsterdam, The Netherlands</conf-loc>, <italic>Proceedings, Part I 14</italic>. (<publisher-loc>Amsterdam, The Netherlands</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>21</fpage>&#x2013;<lpage>37</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Path aggregation network for instance segmentation</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>8759</fpage>&#x2013;<lpage>8768</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Urtasun</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zemel</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Understanding the effective receptive field in deep convolutional neural networks</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>29</volume>, <fpage>4905</fpage>&#x2013;<lpage>4913</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1701.04128</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chi</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Ju</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>YOLO-CWD: A novel model for crop and weed detection based on improved YOLOv8</article-title>. <source>Crop Prot.</source> <volume>192</volume>, <fpage>107169</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cropro.2025.107169</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oerke</surname> <given-names>E. C.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Crop losses to pests</article-title>. <source>J. Agric. Sci.</source> <volume>144</volume>, <fpage>31</fpage>&#x2013;<lpage>43</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/S0021859605005708</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Divvala</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>You only look once: Unified, real-time object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>779</fpage>&#x2013;<lpage>788</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reis</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Kupec</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Daoudi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Real-time flying object detection with YOLOv8</article-title>. <source>arxiv preprint arxiv:2305.09972</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2305.09972</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Faster r-cnn: Towards real-time object detection with region proposal networks</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>39</volume> (<issue>6</issue>), <fpage>1137</fpage>&#x2013;<lpage>1149</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2016.2577031</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sabzi</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Abbaspour-Gilandeh</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Arribas</surname> <given-names>J. I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An automatic visible-range video weed detection, segmentation and classification prototype in potato field</article-title>. <source>Heliyon</source> <volume>6</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.heliyon.2020.e03685</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salazar-Gomez</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Darbyshire</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Sklar</surname> <given-names>E. I.</given-names>
</name>
<name>
<surname>Parsons</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Towards practical object detection for weed spraying in precision agriculture</article-title>. <source>arxiv preprint arxiv:2109.11048</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2109.11048</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Selvaraju</surname> <given-names>R. R.</given-names>
</name>
<name>
<surname>Cogswell</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Das</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Vedantam</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Parikh</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Batra</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Grad-cam: Visual explanations from deep networks via gradient-based localization</article-title>,&#x201d; in <source>Proceedings of the IEEE international conference on computer vision</source> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>618</fpage>&#x2013;<lpage>626</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Guan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xuan</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>GTCBS-YOLOv5s: A lightweight model for weed species identification in paddy fields</article-title>. <source>Comput. Electron. Agric.</source> <volume>215</volume>, <fpage>108461</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108461</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Evaluation of two deep learning-based approaches for detecting weeds growing in cabbage</article-title>. <source>Pest Manage. Sci.</source> <volume>80</volume>, <fpage>2817</fpage>&#x2013;<lpage>2826</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/ps.v80.6</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Sermanet</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Reed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Anguelov</surname> <given-names>D.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). &#x201c;<article-title>Going deeper with convolutions</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Pang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Le</surname> <given-names>Q. V.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Efficientdet: Scalable and efficient object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source> (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>10781</fpage>&#x2013;<lpage>10790</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>He</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Weed identification based on K-means feature learning combined with convolutional neural network</article-title>. <source>Comput. Electron. Agric.</source> <volume>135</volume>, <fpage>63</fpage>&#x2013;<lpage>70</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2017.01.001</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Veeragandham</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Santhi</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>A detailed review on challenges and imperatives of various cnn algorithms in weed detection</article-title>,&#x201d; in <conf-name>2021 International Conference on Artificial Intelligence and Smart Systems (ICAIS)</conf-name> (<publisher-loc>Coimbatore, India</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1068</fpage>&#x2013;<lpage>1073</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veeranampalayam Sivakumar</surname> <given-names>A. N.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Scott</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Psota</surname> <given-names>E. J.</given-names>
</name>
<name>
<surname>Jhala</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Luck</surname> <given-names>J. D.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Comparison of object detection and patch-based classification deep learning models on mid-to late-season weed detection in UAV imagery</article-title>. <source>Remote Sens.</source> <volume>12</volume>, <fpage>2136</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs12132136</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C. Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H. Y. M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source> (<publisher-loc>Vancouver, Canada</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>7464</fpage>&#x2013;<lpage>7475</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Yolov10: Real-time end-to-end object detection</article-title>. <source>arxiv preprint arxiv:2405.14458</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2405.14458</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>He</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Nie</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Gold-YOLO: Efficient object detector via gather-and-distribute Module</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2309.11331</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Niu</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Weed25: A deep learning dataset for weed identification</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>1053329</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.1053329</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A review on weed detection using ground-based machine vision and image processing techniques</article-title>. <source>Comput. Electron. Agric.</source> <volume>158</volume>, <fpage>226</fpage>&#x2013;<lpage>240</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2019.02.005</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Kang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Review of weed detection methods based on computer vision</article-title>. <source>Sensors</source> <volume>21</volume>, <fpage>3647</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s21113647</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Small-target weed-detection model based on YOLO-V4 with improved backbone and neck structures</article-title>. <source>Precis. Agric.</source> <volume>24</volume>, <fpage>2149</fpage>&#x2013;<lpage>2170</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11119-023-10035-7</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Real-time detection and localization of weeds in dictamnus dasycarpus fields for laser-based weeding control</article-title>. <source>Agronomy</source> <volume>14</volume>, <fpage>2363</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agronomy14102363</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>ELA: efficient local attention for deep convolutional neural networks</article-title>. <source>arxiv preprint arxiv:2403.01123</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2403.01123</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>AFPN: Asymptotic feature pyramid network for object detection</article-title>,&#x201d; in <source>2023 IEEE international conference on systems, man, and cybernetics (SMC)</source> (<publisher-loc>Maui, Hawaii</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2184</fpage>&#x2013;<lpage>2189</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>You</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A DNN-based semantic segmentation for detecting weed and crop</article-title>. <source>Comput. Electron. Agric.</source> <volume>178</volume>, <fpage>105750</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105750</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An improved algorithm for sesame seedling and weed detection based on YOLOV7</article-title>. <source>Int. J. Wireless Mobile Computing</source> <volume>26</volume>, <fpage>282</fpage>&#x2013;<lpage>290</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1504/IJWMC.2024.137859</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Inceptionnext: When inception meets convnext</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source> (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>5672</fpage>&#x2013;<lpage>5683</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Lan</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Research on path planning algorithm based on fast target detection</article-title>. <source>J. Artif. Intell. Pract.</source> <volume>7</volume> (<issue>2</issue>). doi:&#xa0;<pub-id pub-id-type="doi">10.23977/jaip.2024.070223</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Real-time vehicle detection based on improved yolo v5</article-title>. <source>Sustainability</source> <volume>14</volume>, <fpage>12274</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/su141912274</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ning</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Design and Testing of an autonomous laser weeding robot for strawberry fields based on DIN-LW-YOLO</article-title>. <source>Comput. Electron. Agric.</source> <volume>229</volume>, <fpage>109808</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109808</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Dang</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). &#x201c;<article-title>Detrs beat yolos on real-time object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name> (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>16965</fpage>&#x2013;<lpage>16974</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>He</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tie</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>W.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Improvement of the YOLOv8 model in the optimization of the weed recognition algorithm in cotton field</article-title>. <source>Plants</source> <volume>13</volume>, <fpage>1843</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/plants13131843</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Vision mamba: Efficient visual representation learning with bidirectional state space model</article-title>. <source>arXiv preprint arXiv:2401.09417</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2401.09417</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>