<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1643700</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>YOLO-lychee-advanced: an optimized detection model for lychee pest damage based on YOLOv11</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Xianjun</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3093154/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Su</surname>
<given-names>Xueping</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Zejie</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xu</surname>
<given-names>Bing</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3149502/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>Guangdong University of Petrochemical Technology</institution>, <addr-line>Maoming</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1369599/overview">Neil Vaughan</ext-link>, University of Exeter, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1937850/overview">Parvathaneni Naga Srinivasu</ext-link>, Amrita Vishwa Vidyapeetham University, India</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/995470/overview">Lokeswari Pinneboyana</ext-link>, Independent researcher, Farmington Michigan, NM, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1484190/overview">Preeta Sharan</ext-link>, The Oxford College of Engineering, India</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Bing Xu, <email xlink:href="mailto:xubing@gdupt.edu.cn">xubing@gdupt.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1643700</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wu, Su, Ma and Xu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wu, Su, Ma and Xu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>We introduce YOLO-Lychee-advanced, a lightweight and high-precision detector for lychee stem-borer damage on fruit surfaces. Built on YOLOv11, the model incorporates (i) a C2f module with dual-branch residual connections to capture fine-grained features of pest holes &#x2264;2 mm, (ii) a CBAM channel-spatial attention block to suppress complex peel-texture interference, and (iii) CIoU loss to tighten bounding-box regression. To mitigate illumination variance, we augment the original 3,061-image dataset to 9,183 samples by simulating direct/back-lighting and adopt a &#x201c;pest-hole only&#x201d; annotation strategy, which improves mAP50&#x2013;95 by 18% over baseline. Experiments conducted on an RTX 3060 with a batch size of 32 and an input size of 416 &#xd7; 416 pixels show YOLO-Lychee-advanced achieves 92.2% precision, 85.4% recall, 91.7% mAP50, and 61.6% mAP50-95, surpassing YOLOv9t and YOLOv10n by 3.4% and 1.7%, respectively, while maintaining 37 FPS real-time speed. Compared with the recent YOLOv9t and YOLOv10n baselines on the same lychee test set, YOLO-Lychee-advanced raises mAP50&#x2013;95 by 3.4% and 1.7%, respectively. Post-processing optimization further boosts precision to 95.5%. A publicly available dataset and PyQt5 visualization tool are provided at <ext-link ext-link-type="uri" xlink:href="https://github.com/Suxueping/Lychee-Pest-Damage-images.git">https://github.com/Suxueping/Lychee-Pest-Damage-images.git</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>lychee stem borer</kwd>
<kwd>object detection</kwd>
<kwd>YOLOv11</kwd>
<kwd>attention mechanism</kwd>
<kwd>data augmentation</kwd>
</kwd-group>
<counts>
<fig-count count="22"/>
<table-count count="14"/>
<equation-count count="13"/>
<ref-count count="49"/>
<page-count count="28"/>
<word-count count="13451"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Lychee(<italic>Litchi chinensis</italic>) is an important fruit crop in tropical and subtropical regions. However, severe infestations of the lychee stem borer(<italic>Conopomorpha sinensis</italic>) can reduce yield by more than 60% and compromise fruit quality. Traditional manual inspections and indiscriminate pesticide application are labor-intensive, error-prone, and contribute to pesticide resistance, underscoring the need for accurate, automated pest detection systems. Manual inspection inefficiency has been highlighted in (<xref ref-type="bibr" rid="B37">Sahu et&#xa0;al., 2023a</xref>; <xref ref-type="bibr" rid="B7">Dang and Wang, 2025</xref>). In recent years, with the rapid development of deep learning technology, computer vision-based object detection technologies have shown great potential in agricultural pest detection (<xref ref-type="bibr" rid="B49">Zhou et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B6">Chen et&#xa0;al., 2025</xref>).</p>
<p>YOLO series models, known for their fast detection capabilities and high accuracy, have been widely applied in object detection tasks. However, existing YOLO models still have shortcomings in small target detection, complex background interference, and positioning accuracy (<xref ref-type="bibr" rid="B47">Xue et&#xa0;al., 2025</xref>). To address these issues, we present an optimized model YOLO-Lychee-advanced based on YOLOv11. We have systematically optimized YOLOv11 for lychee pest detection. Nevertheless, three critical technical gaps persist, preventing existing approaches from attaining truly orchard-deployable performance.</p>
<p>Current YOLO variants lose sub-millimeter pest-hole details after eight-fold down-sampling, misclassify peel textures under variable orchard lighting, and suffer from a scarcity of lychee-specific annotated data, all of which hinder deployment in real orchards.</p>
<p>Our main contributions are:</p>
<list list-type="order">
<list-item>
<p>Introducing the C2f module to enhance feature extraction capabilities, effectively solving the problems of small target detection and complex background interference;</p>
</list-item>
<list-item>
<p>Integrating the CBAM attention mechanism to focus on key features and suppress irrelevant background information;</p>
</list-item>
<list-item>
<p>We augmented the dataset by synthetically generating front- and back-lit variants of each original image, tripling its size to 9183 samples. which enhanced the model&#x2019;s adaptability to complex illumination conditions. A series of experiments were conducted to validate the performance of the YOLO-Lychee-advanced model. The results demonstrated that YOLO-Lychee-advanced outperformed existing YOLO series models in terms of precision, recall, and mean Average Precision (mAP). We provides an effective technical solution for the intelligent detection of lychee diseases and pests.</p>
</list-item>
<list-item>
<p>A Web-based online visualization detection tool for the lychee stem borer, named Lychee Stem Borer Visualization Tool, was designed and implemented. This platform supports dynamic model loading, detection of static images, detection of video streams, and real-time camera detection, thereby providing a convenient tool for lychee disease and pest identification. As a comprehensive end-to-end solution, the platform can be directly deployed and applied in orchard or laboratory settings to assist agricultural technicians and researchers in rapid pest monitoring and decision-making.</p>
</list-item>
</list>
<p>The remainder of the paper reviews related work, presents the methodology, experiments, and conclusions.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<p>This section reviews recent advances in YOLO-based pest detection, highlighting limitations addressed by our method.</p>
<p>In recent years, deep learning technologies have been widely applied in the field of agricultural pest detection. YOLO series models, as representatives of real-time object detection, have been widely used in various object detection tasks due to their fast detection capabilities and high accuracy. YOLO&#x2019;s real-time capability has been validated in agricultural tasks (<xref ref-type="bibr" rid="B28">Ma et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B10">Fu et&#xa0;al., 2024</xref>).YOLOv3 (<xref ref-type="bibr" rid="B34">Redmon and Farhadi, 2018</xref>) introduced multi-scale predictions, YOLOv4 (<xref ref-type="bibr" rid="B4">Bochkovskiy et&#xa0;al., 2020</xref>) consolidated bag-of-freebies and bag-of-specials for speed&#x2013;accuracy trade-offs, YOLOv5 (<xref ref-type="bibr" rid="B29">Nelson. and Solawetz., 2020</xref>) streamlined the training pipeline for production, and YOLOv8 (<xref ref-type="bibr" rid="B11">Gallagher, 2024</xref>) adopted anchor-free heads. These limitations are detailed in Section I.</p>
<p>Mainstream backbones for plant disease detection include ResNet (<xref ref-type="bibr" rid="B17">He et&#xa0;al., 2016</xref>), DenseNet (<xref ref-type="bibr" rid="B43">Tahir et&#xa0;al., 2022</xref>) and Inception (<xref ref-type="bibr" rid="B1">Ali et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B2">Bachhal et&#xa0;al., 2024</xref>), which we use as references for lightweight design.</p>
<p>Lightweight operators&#x2014;depthwise separable convolution (<xref ref-type="bibr" rid="B28">Ma et&#xa0;al., 2023</xref>), attention routing (<xref ref-type="bibr" rid="B10">Fu et&#xa0;al., 2024</xref>), and slim-neck modules (<xref ref-type="bibr" rid="B34">Redmon and Farhadi, 2018</xref>)&#x2014;as well as enlarged receptive field techniques (<xref ref-type="bibr" rid="B4">Bochkovskiy et&#xa0;al., 2020</xref>) have been widely adopted to improve small-target detection efficiency. These strategies inform the design of our C2f module and CBAM integration, yet they were not specifically tailored for sub-millimeter pest-hole features under orchard illumination variation.</p>
<p>There have also been many valuable studies in convolutional modules (<xref ref-type="bibr" rid="B28">Ma et&#xa0;al., 2023</xref>). used depthwise separable convolution to reduce model memory occupancy. DSC decomposes standard convolution into depthwise convolution and pointwise convolution, reducing the number of parameters and computational volume, making the model more suitable for resource-constrained environments (<xref ref-type="bibr" rid="B10">Fu et&#xa0;al., 2024</xref>). introduced the CARAFE upsampling operator to widen the receptive field for data feature fusion. CARAFE uses feature perception recombination to upsample features, predicting recombination kernels for each position based on underlying information and defining recombined features, thereby enhancing the model&#x2019;s ability to capture image details. They also used the C2f Faster structure in the Backbone and Neck of YOLO v8, enhancing the model&#x2019;s feature extraction capabilities. The C2f Faster structure combines partial convolution (PConv) and pointwise convolution (PWConv), reducing the number of parameters and computational complexity while maintaining a certain receptive field range and non-linear representation capabilities. This method is of great reference value in our computational tasks.</p>
<p>In terms of loss functions, <xref ref-type="bibr" rid="B45">Wang et&#xa0;al. (2024)</xref> used the MPDIoU (Minimum Point Distance IoU) loss function for YOLO v8n. The loss function directly predicts the distance between the upper-left and lower-right corners of the predicted bounding box and the actual annotated box, simplifying the comparison of similarity between two bounding boxes and effectively solving the problem of missed detections caused by overlapping fruits, thereby improving detection accuracy. Similarly (<xref ref-type="bibr" rid="B10">Fu et&#xa0;al., 2024</xref>), introduced the Focal SIoU loss function to address the issues of unbalanced positive and negative sample allocation and the limitations of CIoU. Focal SIoU combines Focal Loss and SIoU loss functions, reducing the weight of simple negative samples and enabling the model to focus more on hard-to-classify samples, thereby improving the model&#x2019;s performance when dealing with imbalanced datasets.</p>
<p>In the direction of feature fusion, <xref ref-type="bibr" rid="B26">Lin et&#xa0;al. (2017)</xref> proposed the FPN structure (Feature Pyramid Network), which constructs a feature pyramid to fuse features from different levels, enabling the model to capture both global and local feature information simultaneously. In plant disease detection, this multi-scale feature fusion helps accurately identify different sizes and shapes of disease regions (<xref ref-type="bibr" rid="B25">Li et&#xa0;al., 2022b</xref>; <xref ref-type="bibr" rid="B12">Ghayoumi, 2024</xref>), especially suitable for processing high-resolution agricultural images. This helps improve the accuracy and robustness of disease detection. PANet (<xref ref-type="bibr" rid="B27">Liu et&#xa0;al., 2018</xref>)further optimizes the feature fusion path based on FPN, improving feature propagation efficiency and model performance through bidirectional feature fusion. BoTNet&#x2019;s (<xref ref-type="bibr" rid="B41">Srinivas et&#xa0;al., 2021</xref>) MHSA module can handle feature maps of different scales, enabling the model to better capture global information and local details of targets, improving the model&#x2019;s recognition capabilities in complex backgrounds and occlusion situations (<xref ref-type="bibr" rid="B28">Ma et&#xa0;al., 2023</xref>).</p>
<p>Attention mechanisms enhance the model&#x2019;s focus on key features by automatically learning important regions in images. For example, in plant disease detection (<xref ref-type="bibr" rid="B33">Ramamurthy et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B14">Guo et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B16">Han et&#xa0;al., 2024</xref>), attention mechanisms can help the model more accurately locate disease regions, thereby improving detection accuracy. This is especially useful when disease features are small or not obvious. This helps in early detection of diseases, allowing timely measures to be taken to reduce losses. Related work includes Guo et&#xa0;al (<xref ref-type="bibr" rid="B13">Guo et&#xa0;al., 2022</xref>), who introduced attention mechanisms such as SE, ECA, and CBAM into target detection models like Faster R-CNN, YOLOx, and SSD, significantly improving the detection accuracy of grape leaf diseases and model operational efficiency. <xref ref-type="bibr" rid="B24">Li et&#xa0;al. (2022a)</xref> introduced attention mechanisms such as scSE and CA into the backbone network, enabling the improved network to more quickly and accurately identify and locate defect regions, with stronger generalization capabilities for defect categories and significantly improved image defect detection accuracy. SENet (<xref ref-type="bibr" rid="B21">Hu et&#xa0;al., 2018</xref>) enhances the model&#x2019;s expression of key features by adding channel attention modules between convolutional layers, dynamically adjusting the importance weights of each channel (<xref ref-type="bibr" rid="B10">Fu et&#xa0;al., 2024</xref>). introduced the BiFormer attention mechanism to focus adaptively on small area features, improving the model&#x2019;s detection accuracy for small targets (<xref ref-type="bibr" rid="B34">Redmon and Farhadi, 2018</xref>). introduced the CBAM attention mechanism, combining channel attention and spatial attention to enhance the model&#x2019;s feature extraction capabilities, reducing background interference and improving model robustness.</p>
<p>MobileNet, through the use of depthwise separable convolution, significantly reduces the model&#x2019;s parameter count and computational volume, making it more suitable for mobile devices. This is very important for practical applications in plant disease detection, as many detection tasks need to be performed in the field in real-time (<xref ref-type="bibr" rid="B35">Ridnik et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B38">Sangjan et&#xa0;al., 2021</xref>). Other lightweight designs include (<xref ref-type="bibr" rid="B28">Ma et&#xa0;al., 2023</xref>), who used depthwise separable convolution to significantly reduce the model&#x2019;s parameter count and computational volume, making the model more suitable for mobile and embedded devices (<xref ref-type="bibr" rid="B34">Redmon and Farhadi, 2018</xref>). improved the detection of strawberries and peduncles through the lightweight design of the Slim-neck module, reducing the model&#x2019;s computational complexity and improving operational efficiency, further promoting the deployment of models in practical applications.</p>
<p>In terms of model optimization through knowledge distillation and pruning, <xref ref-type="bibr" rid="B18">Hinton et&#xa0;al. (2015)</xref> transferred the knowledge of large complex models to smaller models (<xref ref-type="bibr" rid="B19">Hoang and Jo, 2021</xref>; <xref ref-type="bibr" rid="B32">Qian et&#xa0;al., 2021</xref>). Knowledge distillation can significantly reduce the model&#x2019;s parameter count and computational volume without significantly reducing performance. In the agricultural field, knowledge distillation can be used to develop lightweight models that can run on resource-constrained devices such as smartphones and embedded systems. <xref ref-type="bibr" rid="B15">Han et&#xa0;al. (2015)</xref> used pruning techniques to remove unimportant weights or neurons from the model, further compressing the model size and improving operational efficiency (<xref ref-type="bibr" rid="B10">Fu et&#xa0;al., 2024</xref>). optimized the model by using the C2f Faster structure and CARAFE upsampling operator, reducing the model&#x2019;s parameter count and computational volume while maintaining high detection accuracy. In agricultural image analysis (<xref ref-type="bibr" rid="B8">Ekanayake et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B31">Park and Kim, 2022</xref>), model pruning can reduce the demand for computational resources, enabling the model to process image data faster and improve detection real-time performance.</p>
<p>Data augmentation techniques (such as rotation, flipping, cropping, and color adjustment) (<xref ref-type="bibr" rid="B36">Ryo, 2022</xref>; <xref ref-type="bibr" rid="B40">Shoaib et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B22">Isinkaye et&#xa0;al., 2024</xref>) can increase the diversity of training data and improve the model&#x2019;s generalization ability. In agricultural image data, data augmentation can simulate different lighting conditions, shooting angles, and disease development stages, thereby improving the model&#x2019;s robustness in practical applications.</p>
<p>
<xref ref-type="bibr" rid="B5">Chen et&#xa0;al. (2020)</xref> used pre-trained models (such as IMa, HgeNet pre-trained models) trained on large-scale datasets and fine-tuned them through transfer learning, significantly improving model performance in plant disease detection tasks. In plant disease detection, pre-trained models can quickly adapt to new disease types and image features, reducing the workload of data annotation and training time. For example, models such as VGG16, Inception V3, and ResNet50 have been fine-tuned through transfer learning (<xref ref-type="bibr" rid="B3">Bhatti et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B37">Sahu et&#xa0;al., 2023a</xref>).</p>
</sec>
<sec id="s3">
<label>3</label>
<title>YOLOv11 overview</title>
<p>YOLOv11 is an important variant of the YOLO series, optimized for real-time object detection tasks. As a representative of advanced single-stage object detection models, YOLOv11 has demonstrated significant technological advantages and application potential in agricultural visual tasks. In the realm of crop health monitoring, this model is capable of efficiently processing complex agricultural scene images and accurately identifying a variety of crop abnormal states, including key agricultural information such as disease characteristics, pest traces, and symptoms of nutrient deficiency. The unique lightweight network structure and multiscale feature fusion mechanism of YOLOv11 enable it to maintain high detection accuracy while adapting to the practical challenges of variable target scales and complex backgrounds commonly encountered in agricultural scenarios. Compared with traditional detection algorithms, YOLOv11 has shown marked improvements in feature extraction capabilities, detection accuracy for small targets, and model generalizability. The experimental results confirmed that the integration of the CBAM attention mechanism and CIoU loss function led to a definitive performance breakthrough, with the model attaining a mAP50&#x2013;95 of 61.6% (95% CI: 60.1&#x2013;63.1%). These enhancements provide an ideal framework for developing high-precision intelligent agricultural monitoring systems (<xref ref-type="bibr" rid="B39">Shen et&#xa0;al., 2025</xref>; <xref ref-type="bibr" rid="B46">Wang K. et&#xa0;al., 2025</xref>).</p>
<p>Detailed architectural parameters are presented in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> after the model improvements.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Implementation details.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Parameter/item</th>
<th valign="middle" align="left">Value/specification</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Input resolution</td>
<td valign="middle" align="left">416&#xd7;416 pixels</td>
</tr>
<tr>
<td valign="middle" align="left">Optimizer</td>
<td valign="middle" align="left">SGD (momentum0.937,<break/>Weight decay=5&#xd7;10<sup>&#x2212;4</sup>)</td>
</tr>
<tr>
<td valign="middle" align="left">Kernel Sizes</td>
<td valign="middle" align="left">3&#xd7;3,1&#xd7;1,7&#xd7;7 (CBAM)</td>
</tr>
<tr>
<td valign="middle" align="left">Stride</td>
<td valign="middle" align="left">1or2 (stage-dependent)</td>
</tr>
<tr>
<td valign="middle" align="left">Activation Function</td>
<td valign="middle" align="left">SiLU (Swish)</td>
</tr>
<tr>
<td valign="middle" align="left">Batch Size</td>
<td valign="middle" align="left">32</td>
</tr>
<tr>
<td valign="middle" align="left">Epochs</td>
<td valign="middle" align="left">200</td>
</tr>
<tr>
<td valign="middle" align="left">Initial Learning Rate</td>
<td valign="middle" align="left">Cosine decay 1&#xd7;10<sup>&#x2212;3</sup> to 1&#xd7;10<sup>&#x2212;4</sup>
</td>
</tr>
<tr>
<td valign="middle" align="left">Hardware</td>
<td valign="middle" align="left">NVIDIA RTX 3060 12GB,CUDA12.4</td>
</tr>
<tr>
<td valign="middle" align="left">Software</td>
<td valign="middle" align="left">PyTorch1.13,Python3.8,<break/>Ubuntu20.04</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3_1">
<label>3.1</label>
<title>YOLO-lychee-advanced architecture details</title>
<p>Therefore, research on the improvement of the YOLOv11 model can further enhance the accuracy and reliability of agricultural image analysis, we have considered combining the C2f module and the CBAM attention mechanism, which significantly enhances feature extraction and the detection capability for small targets. The dual-branch C2f module captures richer features and multiple convolutional operations, reducing false positives and false negatives. Meanwhile, the CBAM attention mechanism optimizes features from both the channel and spatial dimensions, focusing on key regions and suppressing background interference, thereby improving the model&#x2019;s detection performance in complex scenes.</p>
<p>Its core architecture consists of a backbone network, neck network, and head network, achieving efficient detection through multi-scale feature fusion (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>YOLOv11 model structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g001.tif">
<alt-text content-type="machine-generated">Diagram of a neural network architecture divided into Backbone, Neck, and Head sections. The Backbone contains multiple convolutional layers and blocks labeled Conv, C3k2, SPPF, and C2PSA. The Neck section features concatenation, upsampling, and additional convolutional operations. The Head is labeled Detect. The flow of information is shown with arrows connecting the components.</alt-text>
</graphic>
</fig>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Backbone network</title>
<p>Function: To extract image features layer by layer and generate feature maps of different scales.</p>
<p>Core Module C3k2 (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Structure of the C3k2 module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g002.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a neural network module named &#x201c;C3k2&#x201d;. It includes blocks labeled &#x201c;Conv&#x201d; for convolution, &#x201c;Split&#x201d;, &#x201c;Concat&#x201d;, &#x201c;C3k&#x201d;, and an unspecified output. Arrows indicate data flow between them, with a focus on processing and concatenation steps.</alt-text>
</graphic>
</fig>
<p>Structure:</p>
<p>The input feature map passes through multiple convolutional layers (Conv + BN + SiLU activation).</p>
<p>Residual connections add the input directly to the output, alleviating the gradient vanishing problem.</p>
<p>The output feature map is passed to subsequent layers, gradually expanding the receptive field.</p>
<p>Mathematical Expression:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Role:</p>
<p>By stacking multiple convolutional layers (with BN and activation functions), it extracts features of different scales, enhancing feature representation capabilities. The residual structure directly adds the input and output features, effectively alleviating the gradient vanishing problem in deep networks and stabilizing feature propagation. In the backbone network, this module expands the receptive field progressively, capturing global context information. Ultimately, in the subsequent feature fusion stage, it optimizes the fused features from multiple levels, significantly enhancing feature discriminability and providing a high-quality feature base for accurate detection.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Neck network</title>
<p>Function: To fuse the multi-scale features output by the backbone network and provide more expressive features for the head network.</p>
<p>Design Features:</p>
<p>Upsampling: The high-level feature map (e.g., 104&#xd7;104) is upsampled to the same resolution as the low-level feature (e.g., 208&#xd7;208).</p>
<p>Feature concatenation: The upsampled features are concatenated with the corresponding features from the backbone network, enhancing detail information.</p>
</sec>
<sec id="s3_1_3">
<label>3.1.3</label>
<title>Head network</title>
<p>Function: To output detection results using the fused features from the neck network.</p>
<p>Design Features:</p>
<p>Detection head: Predicts target positions and categories through anchor mechanisms.</p>
<p>Detailed architectural parameters are presented in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> after the model improvements.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Evaluation metrics</title>
<p>Model performance evaluation is a core aspect of object detection tasks. To objectively quantify the performance of the proposed model, our model adopts precision (P), recall (R), mean average precision (mAP), and mAP50&#x2013;95 as the primary evaluation metrics (<xref ref-type="bibr" rid="B9">Everingham et&#xa0;al., 2010</xref>; <xref ref-type="bibr" rid="B20">Hosang et&#xa0;al., 2016</xref>), defined as follows:</p>
<p>Precision (P): Reflects the proportion of samples predicted as positive that are truly positive, calculated as:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtext>P</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>FP</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where TP (True Positive) is the number of true positives, and FP (False Positive) is the number of false positives.</p>
<p>Recall (R): Reflects the proportion of truly positive samples that are correctly predicted by the model, calculated as:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mtext>R</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where FN (False Negative) is the number of false negatives.</p>
<p>Mean Average Precision (mAP): First, the average precision (AP) for a single category is calculated as the area under the precision-recall curve (PR curve):</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtext>AP</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mrow><mml:mn>1</mml:mn></mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mtext>P</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>R</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mtext>dR</mml:mtext>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Then, the mAP is obtained by averaging the APs of all categories:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mtext>mAP</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mtext>N</mml:mtext>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where N is the total number of detection categories.</p>
<p>1) mAP50-95: The mAP is calculated for each IoU threshold from 0.5 to 0.95 (with a step size of 0.05), and the average of these mAP values is taken to comprehensively evaluate the model&#x2019;s robustness under different localization accuracy requirements.</p>
<p>To facilitate reproducibility, the four cases in the confusion matrix are defined as follows:</p>
<list list-type="bullet">
<list-item>
<p>TP: A predicted bounding box has IoU &#x2265; 0.5 with a ground-truth insect-hole box and the predicted class is &#x201c;insect_pest&#x201d;.</p>
</list-item>
<list-item>
<p>FP: A predicted box has no matching ground-truth box with IoU &#x2265; 0.5, or the matched box belongs to a different class.</p>
</list-item>
<list-item>
<p>FN: A ground-truth insect-hole box has no predicted box with IoU &#x2265; 0.5.</p>
</list-item>
<list-item>
<p>TN: Regions of lychee surface without any ground-truth insect-hole and where the model produces no detections.</p>
</list-item>
</list>
<p>Because the task is single-class, TNs are not involved in mAP but are considered when quantifying background false alarms (FP).</p>
<p>Priority Explanation: In object detection systems, mAP50&#x2013;95 is the most comprehensive due to its coverage of multiple IoU thresholds and is prioritized as the core metric. mAP, precision, and recall are used as auxiliary analysis bases. This design avoids potential evaluation biases introduced by a single IoU threshold (e.g., mAP50).</p>
</sec>
<sec id="s5">
<label>5</label>
<title>Experimental data and processing optimization</title>
<sec id="s5_1">
<label>5.1</label>
<title>Experimental data collection and processing</title>
<p>Our study adopts a single-centre, prospective laboratory design to evaluate the detection accuracy of the proposed YOLO-Lychee-advanced model for lychee stem-borer damage under controlled indoor conditions.</p>
<p>The detailed in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> (<xref ref-type="bibr" rid="B42">Srinivasu et&#xa0;al., 2025</xref>).</p>
<p>Our lychee pest dataset is novel and scientifically valuable. The lychee stem borer&#x2014;the primary threat to fruit quality and yield&#x2014;causes internal rot and premature drop; severe infestations can reduce yield by more than 60%. Chemical control can also lead to pesticide residue risks. Therefore, solving the detection problem of this pest is crucial for the development of the lychee industry.</p>
<p>The experimental team collected lychee samples from the core production area in Maoming, Guangdong, and brought them back to the laboratory. Using high-precision imaging equipment(<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>), they focused on the lychee stem borer and captured images of multiple varieties, including Guiwei and Feizixiao, from different angles and at different pest infestation levels (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>). The shooting process was based on natural indoor lighting, although the shooting background was not completely uniform and simple, it truly reflected the actual state of lychee pest infestation. A total of 3061 images were collected, providing rich and reliable first-hand data for the study.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Configuration and default settings of image acquisition devices.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Parameter/model</th>
<th valign="middle" align="center">iPhone 12</th>
<th valign="middle" align="center">Honor 50</th>
<th valign="middle" align="center">Honor X 50</th>
<th valign="middle" align="center">real me GT neo(speed edition)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Primary Sensor Model</td>
<td valign="middle" align="center">Apple Custom</td>
<td valign="middle" align="center">Samsung HM 2(108MP)</td>
<td valign="middle" align="center">Samsung HM 6(108MP)</td>
<td valign="middle" align="center">Sony IMX 682(64MP)</td>
</tr>
<tr>
<td valign="middle" align="center">Effective Resolution</td>
<td valign="middle" align="center">12MP(Default)</td>
<td valign="middle" align="center">12MP(9-in-1binning) 108MP(Native)</td>
<td valign="middle" align="center">12MP(9-in-1binning) 108MP(Native)</td>
<td valign="middle" align="center">16MP(4-in-lbinning) 64MP(Native)</td>
</tr>
<tr>
<td valign="middle" align="center">Default OutputResolution</td>
<td valign="middle" align="center">4032x3024px</td>
<td valign="middle" align="center">4000x3000px(binned)</td>
<td valign="middle" align="center">4000&#xd7;3000px(binned)</td>
<td valign="middle" align="center">4624x34683px(binned)</td>
</tr>
<tr>
<td valign="middle" align="center">Native High-Res Mode</td>
<td valign="middle" align="center">Not supported</td>
<td valign="middle" align="center">12032x9024px</td>
<td valign="middle" align="center">12000x9000px</td>
<td valign="middle" align="center">9280x6944px</td>
</tr>
<tr>
<td valign="middle" align="center">Sensor Size(inch)</td>
<td valign="middle" align="center">1/2.55*</td>
<td valign="middle" align="center">1/1.52&#x201d;</td>
<td valign="middle" align="center">1/1.67&#x201d;</td>
<td valign="middle" align="center">1/1.73&#x201d;</td>
</tr>
<tr>
<td valign="middle" align="center">PixelSize(um)</td>
<td valign="middle" align="center">1.4(Native)</td>
<td valign="middle" align="center">2.1(binned)</td>
<td valign="middle" align="center">1.92(binned)</td>
<td valign="middle" align="center">1.6(binned)</td>
</tr>
<tr>
<td valign="middle" align="center">Aperture(f)</td>
<td valign="middle" align="center">f/1.6</td>
<td valign="middle" align="center">f/1.9</td>
<td valign="middle" align="center">f/1.75</td>
<td valign="middle" align="center">f/1.8</td>
</tr>
<tr>
<td valign="middle" align="center">Key Features</td>
<td valign="middle" align="center">Smart HDR 3.Deep Fusion</td>
<td valign="middle" align="center">Multi-frame A IEnhancement</td>
<td valign="middle" align="center">Multi-frame NoiseReduction</td>
<td valign="middle" align="center">A I SceneDetection</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Lychee fruits and interiors affected by lychee stem borer.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g003.tif">
<alt-text content-type="machine-generated">Three images of infested fruit showing worms on the surface and inside. The images reveal worms on the bumpy red skin and within the fruit's flesh, indicating contamination.</alt-text>
</graphic>
</fig>
<sec id="s5_1_1">
<label>5.1.1</label>
<title>Implementation platform details</title>
<p>All model training and evaluation were performed on a standardized local workstation to ensure reproducibility and fair comparison. The hardware and software configurations are listed in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Implementation environment configuration.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Component</th>
<th valign="middle" align="left">Specification</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">CPU</td>
<td valign="middle" align="left">Intel Corei7-12700KF @3.6GHz base</td>
</tr>
<tr>
<td valign="middle" align="left">GPU</td>
<td valign="middle" align="left">NVIDIA GeForce RTX 3060&#x2013;12 GB</td>
</tr>
<tr>
<td valign="middle" align="left">RAM</td>
<td valign="middle" align="left">32 GB DDR4-3200</td>
</tr>
<tr>
<td valign="middle" align="left">OS</td>
<td valign="middle" align="left">Ubuntu 22.04 LTS</td>
</tr>
<tr>
<td valign="middle" align="left">CUDA/cuDNN</td>
<td valign="middle" align="left">12.4/8.6</td>
</tr>
<tr>
<td valign="middle" align="left">Python</td>
<td valign="middle" align="left">3.8</td>
</tr>
<tr>
<td valign="middle" align="left">PyTorch</td>
<td valign="middle" align="left">1.13.1</td>
</tr>
<tr>
<td valign="middle" align="left">YOLO Framework</td>
<td valign="middle" align="left">Ultralytics YOLOv11n<break/>(YOLOv11n.ptpre-trained)</td>
</tr>
<tr>
<td valign="middle" align="left">Batch Size</td>
<td valign="middle" align="left">32</td>
</tr>
<tr>
<td valign="middle" align="left">Image Size</td>
<td valign="middle" align="left">416&#xd7;416px</td>
</tr>
<tr>
<td valign="middle" align="left">Epochs</td>
<td valign="middle" align="left">200</td>
</tr>
<tr>
<td valign="middle" align="left">Optimizer</td>
<td valign="middle" align="left">SGD(momentum0.937)</td>
</tr>
<tr>
<td valign="middle" align="left">Learning Rate</td>
<td valign="middle" align="left">0.001(cosinedecay)</td>
</tr>
<tr>
<td valign="middle" align="left">Weight Decay</td>
<td valign="middle" align="left">0.0005</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Ethics statement: This study did not involve any human or vertebrate subjects, and all lychee fruits were commercially purchased surplus samples.</p>
</sec>
<sec id="s5_1_2">
<label>5.1.2</label>
<title>Dataset composition and class statistics</title>
<p>The dataset originates from a commercial lychee orchard in Maoming, Guangdong, China. We augmented the original 3,061 images to 9,183 by simulating direct and back-lighting conditions (Section V.D). All images were manually annotated under the &#x201c;Only pest holes&#x201d; strategy (Section V.B).</p>
<p>To ensure a 95% confidence interval width &#x2264; 5% for mAP50&#x2013;95 at an expected value of 0.60, we calculated that at least 3&#x2013;061 original images were required (PASS 16.0, two-sided &#x3b1; = 0.05, power = 0.90). After 3-fold illumination augmentation (see Section V.D), the final dataset comprised 9&#x2013;183 images, preserving the same CI width while accounting for the 70/20/10 split. The number of instances per class is shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Number of instances per class.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Class name</th>
<th valign="middle" align="left">Training set</th>
<th valign="middle" align="left">Validation set</th>
<th valign="middle" align="left">Testset</th>
<th valign="middle" align="left">Total</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">insect_pest</td>
<td valign="middle" align="left">6,428</td>
<td valign="middle" align="left">1,836</td>
<td valign="middle" align="left">919</td>
<td valign="middle" align="left">9,183</td>
</tr>
<tr>
<td valign="middle" align="left">Normal</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">0</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>This is a single-class detection task targeting lychee stem borer damage (pest holes). Training set (6,428 images) is augmented to 9,183 to improve single-class robustness, following common practice of using thousands rather than hundreds of samples for deep-learning detection tasks.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Comparison of annotation strategies</title>
<p>In the field of lychee pest detection, the choice of annotation strategy and the model&#x2019;s learning performance under different datasets are crucial for improving detection accuracy and efficiency. We deeply compared two annotation strategies and analyzed the training and validation loss curves under corresponding datasets in detail, aiming to provide a solid basis for subsequent model optimization and dataset processing (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Annotation situations of data images.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g004.tif">
<alt-text content-type="machine-generated">At the top is a set of 4*4 local pest images, which branch into two groups labeled "Strategy 1" and "Strategy 2." "Strategy 1" involves more refined annotation of wormholes, focusing solely on the wormhole areas, while "Strategy 2" includes annotations of both the wormholes and the surrounding peel regions.</alt-text>
</graphic>
</fig>
<p>Strategy 1: Annotate the pest holes and a small amount of peel area (providing spatial context information).</p>
<p>Strategy 2: Annotate only the core area of the pest holes (focusing on subtle features).</p>
<p>The original dataset consisted of 3061 images. Trained under the configuration specified in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The experimental results are as follows (<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>):</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Comparison of the effects of two annotation methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Indicator group</th>
<th valign="middle" align="center">Training set</th>
<th valign="middle" align="center">Validation set</th>
<th valign="middle" align="center">Test set</th>
<th valign="middle" align="center">P(%)</th>
<th valign="middle" align="center">R(%)</th>
<th valign="middle" align="center">mAP 50(%)</th>
<th valign="middle" align="center">mAP 50-95(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Only Wormholes</td>
<td valign="middle" align="center">2143</td>
<td valign="middle" align="center">612</td>
<td valign="middle" align="center">306</td>
<td valign="middle" align="center">89.6</td>
<td valign="middle" align="center">78.4</td>
<td valign="middle" align="center">87.7</td>
<td valign="middle" align="center">52.4</td>
</tr>
<tr>
<td valign="middle" align="center">Small Amount of Fruit<break/>Peel+Wormholes</td>
<td valign="middle" align="center">2143</td>
<td valign="middle" align="center">612</td>
<td valign="middle" align="center">306</td>
<td valign="middle" align="center">94</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">92.7</td>
<td valign="middle" align="center">44.4</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Strategy 1: &#x201c;Small Amount of Peel + Pest Hole&#x201d; Annotation</p>
<p>This annotation strategy annotates both the pest hole and a small amount of peel area, allowing the model to establish a strong association between pest damage and peel texture and color during training. In the orchard pest distribution statistics scenario, this association plays a significant role, with the model achieving a precision (P) of 82.1% and recall (R) of 78.3%. This indicates that the strategy effectively covers various abnormal features on the fruit surface, providing reliable data support for a comprehensive understanding of orchard pest distribution.</p>
<p>However, this strategy also has certain limitations. The annotation of non-pest areas (i.e., the small amount of peel) introduces additional noise, limiting the model&#x2019;s performance in the mAP50&#x2013;95 metric. This means that in precisely capturing pest hole boundaries and identifying minor lesions, the model still has significant room for improvement.</p>
<p>Strategy 2: &#x201c;Pest Hole Only&#x201d; Annotation</p>
<p>This strategy focuses strictly on the core area of the pest holes. In the early stages of training, the model&#x2019;s precision (79.6%) and recall (75.2%) under this strategy were slightly lower than those of Strategy 1. However, the mAP50&#x2013;95 metric saw a significant improvement, increasing from 44.4% to 52.4%, a rise of 18.0%.</p>
<p>This significant improvement is due to the fact that this strategy forces the model to focus on the essential features of the pest damage, reducing interference from non-related areas (such as the peel). The experimental results fully demonstrate that this annotation method is more conducive to high-precision localization in robotic harvesting systems, enabling more accurate identification and localization of pest holes and providing more reliable guidance for subsequent harvesting and processing tasks (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Comparison of the effects of two annotation methods.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g005.tif">
<alt-text content-type="machine-generated">Bar chart titled &#x201c;Performance Comparison&#x201d; shows metric values for two dataset compositions: &#x201c;Only Wormholes&#x201d; and &#x201c;Wormholes + Fruit Peel.&#x201d; Metrics include Precision, Recall, mAP50, and mAP50-95, represented in blue, green, pink, and yellow respectively. &#x201c;Only Wormholes&#x201d; scores are 89.6 (Precision), 78.4 (Recall), 87.7 (mAP50), and 52.4 (mAP50-95). &#x201c;Wormholes + Fruit Peel&#x201d; scores are 94.0 (Precision), 89.1 (Recall), 92.7 (mAP50), and 44.4 (mAP50-95).</alt-text>
</graphic>
</fig>
<p>Taking into account the pros and cons of both strategies, we ultimately selected Strategy 2, which focuses solely on the pest holes, as the basis for subsequent research. To compensate for its lower recall, data augmentation techniques are planned to be employed for further optimization.</p>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Analysis of loss curves for different datasets</title>
<p>To gain a deeper understanding of the learning characteristics and performance of the model under different annotation strategies, we conducted a detailed comparison of the loss curves&#xa0;for the &#x201c;Small Amount of Fruit Peel + pest holes&#x201d; and &#x201c;Only pest holes&#x201d; datasets. This analysis covered the training bounding box loss, training classification loss, training distribution focusing loss, and the corresponding validation loss curves (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>).</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Analysis of loss curves for lychee pest detection models: &#x201c;small amount of peel + pest hole&#x201d; vs. &#x201c;pest hole only&#x201d; datasets.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g006.tif">
<alt-text content-type="machine-generated">Six loss curves show training and validation loss for a model. Top row: Training Box Loss, Cls Loss, and Dfl Loss curves. Bottom row: Validation Box Loss, Cls Loss, and Dfl Loss curves. Two datasets, &#x201c;Small Amount of Fruit Peel+Wormholes&#x201d; and &#x201c;Only Wormholes&#x201d;, are color-coded in blue and green, respectively. Curves indicate loss decreases over 200 epochs.</alt-text>
</graphic>
</fig>
<sec id="s5_3_1">
<label>5.3.1</label>
<title>Explanation of loss function formulas</title>
<sec id="s5_3_1_1">
<label>5.3.1.1</label>
<title>Box loss (bounding box localization loss)</title>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>DFL Loss (Dynamic Distribution Loss):</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mi>:</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<disp-formula>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:msub>
<mml:mtext>w</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mi>:</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the true label value (0 or 1), <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted probability of the positive class. The formula is based on the idea of cross-entropy loss, with dynamic weights applied to the cross-entropy losses of different samples to highlight the role of samples with larger errors in the loss calculation.</p>
</sec>
<sec id="s5_3_1_2">
<label>5.3.1.2</label>
<title>Classification loss</title>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Poly Loss:</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the category weight generated by meta-learning (to address class imbalance). <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>1.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x3b3;</mml:mtext>
<mml:mo>=</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.5:Suppressing the gradients of easily classified samples</p>
<p>3) DFL Loss (Distribution Focusing Loss):</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mtext>Formula</mml:mtext>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0.2</mml:mn>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The DFL loss is independently monitored (with a weight of 0.2) to reflect the stability of the distribution learning of bounding boxes. If the curve fluctuates significantly, the target distribution parameter settings (such as the number of bins) should be checked.</p>
</sec>
<sec id="s5_3_1_3">
<label>5.3.1.3</label>
<title>Loss curve analysis</title>
<p>1) Training Loss Curve Comparison</p>
<p>&#x2022; Training Bounding Box Loss Curve</p>
<p>The downward trend of the blue curve corresponding to the &#x201c;Only pest holes&#x201d; dataset is relatively smoother. This indicates that the model&#x2019;s optimization process is more stable when learning to localize bounding boxes containing only pest hole targets. In contrast, the green curve representing the &#x201c;Small Amount of Fruit Peel + pest holes&#x201d; dataset, although also showing an overall decline, exhibits a fluctuation amplitude of 12-15% in the early stage (epochs 0-50). This suggests that the introduction of a small amount of fruit peel leads to gradient instability in two phases: first, a noticeable loss rebound occurs around epoch 30 (with an approximate 8% increase), followed by a gradual stabilization after epoch 50. This two-stage convergence pattern reveals that the model needs to first overcome the interference of fruit peel features before effectively learning bounding box localization.</p>
<p>&#x2022; Training Classification Loss Curve</p>
<p>Both curves exhibit a rapid downward trend in the early stages of training (a decline of approximately 60% within the first 20 epochs), demonstrating the model&#x2019;s strong initial learning capacity for classification features. Notably, the green curve displays minor fluctuations (with an amplitude of approximately 5%) between epochs 40 and 60, coinciding with the period of fluctuation in bounding box loss. This suggests that the fruit peel features temporarily interfere with the model&#x2019;s multitask learning. As the number of training epochs increases, the difference in the final convergence values of the two curves is less than 3%, indicating that after sufficient training, the model is capable of essentially overcoming the classification interference caused by the fruit peel features.</p>
<p>&#x2022; Training Distribution Focusing Loss Curve</p>
<p>The blue curve corresponding to the &#x201c;Only pest holes&#x201d; dataset exhibits a monotonic decline (with a total reduction of 75%), indicating the model&#x2019;s highly efficient learning of the distribution features of pure pest hole targets. In contrast, the green curve representing the dataset containing fruit peel displays three distinct characteristics (<xref ref-type="bibr" rid="B34">Redmon and Farhadi, 2018</xref>): a slow initial decline (only a 30% reduction in the first 30 epochs) (<xref ref-type="bibr" rid="B4">Bochkovskiy et&#xa0;al., 2020</xref>), periodic fluctuations in the middle stage (epochs 30-100, with a period of approximately 15 epochs and an amplitude of 8%), and (<xref ref-type="bibr" rid="B11">Gallagher, 2024</xref>) a persistent loss difference of 0.02-0.03 after epoch 150. These tripartite characteristics clearly demonstrate that the fruit peel not only delays the learning progress of the distribution features but also continuously affects the model&#x2019;s precision in modeling the target probability distribution.</p>
<p>2) Validation Loss Curve Comparison</p>
<p>&#x2022; Validation Bounding Box Loss Curve</p>
<p>The blue curve exhibits ideal convergence characteristics: the gap between the validation loss and the training loss remains stable within 0.01, indicating that the model possesses good generalization ability. In contrast, the green curve reveals three issues: the validation loss is consistently higher than the training loss (with an average difference of 0.05), two significant peaks appear at epochs 75 and 125 (increasing by 22% and 18%, respectively), and the final stable value is 35% higher than that of the blue curve. This tripartite phenomenon of &#x201c;high baseline-strong fluctuation-large gap&#x201d; directly reflects the localization performance degradation caused by the fruit peel: the model&#x2019;s localization accuracy for samples containing fruit peel is not only lower but also unstable.</p>
<p>&#x2022; Validation Classification Loss Curve</p>
<p>After epoch 50, the two curves are essentially parallel, but the green curve is offset by approximately 0.015. Further analysis reveals that this offset primarily originates from the persistent misclassification of two types of samples: pest holes partially obscured by fruit peel (accounting for 63% of the misclassified samples) and irregularly shaped dried fruit peel (37%). Notably, after epoch 100, the fluctuation coefficient (standard deviation/mean) of the green curve is 40% higher than that of the blue curve, indicating that even though the overall trend is stable, the presence of fruit peel still introduces greater uncertainty in classification predictions.</p>
<p>&#x2022; Validation Distribution Focusing Loss Curve</p>
<p>The blue curve exhibits a typical exponential decay (R&#xb2; = 0.93), while the green curve is best fitted by a linear decline (R&#xb2; = 0.81) superimposed with sinusoidal fluctuations (amplitude 0.008, period 25 epochs). This difference in mathematical characteristics holds significant implications: the pure pest hole data enable the model to stably optimize its distribution predictions, whereas the presence of fruit peel introduces periodic interference&#x2014;likely due to a random fluctuation of approximately 15% in the proportion of fruit peel across different batches in the validation set, causing the model to oscillate between focusing on pest hole features and adapting to the interference from fruit peel.</p>
</sec>
<sec id="s5_3_1_4">
<label>5.3.1.4</label>
<title>Summary</title>
<p>Overall, the &#x201c;Only pest holes&#x201d; dataset shows better convergence and stability in the model&#x2019;s training and validation processes, especially in bounding box localization and target distribution feature learning. In contrast, the &#x201c;Small Amount of Fruit Peel + pest holes&#x201d; dataset presents certain challenges in some loss optimization processes due to the interference of peel factors.</p>
<p>These comparative results provide important references for subsequent model optimization and dataset processing. Based on this, we selects the &#x201c;Pest Hole Only&#x201d; annotation strategy as the basis for subsequent research and plans to combine data augmentation techniques to compensate for its lower recall. Future research can further explore how to better handle the interference caused by peel factors and how to optimize model structure and training methods to further improve the performance of lychee pest detection models to meet the needs of practical applications.</p>
</sec>
</sec>
</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Data augmentation strategy</title>
<sec id="s6_1">
<label>6.1</label>
<title>Background and motivation for data augmentation</title>
<p>To address the potential overfitting problem caused by limited training data and to enhance the model&#x2019;s adaptability to complex real-world scenarios, we employed data augmentation techniques based on simulated lighting conditions to systematically expand the original lychee pest dataset.</p>
<p>In actual orchard collection scenarios, lighting conditions are complex and variable. Within the same time period, the different lighting angles on lychee fruits can lead to significant differences in the appearance of pest damage under direct and backlighting conditions. To simulate these real-world scenarios and improve the model&#x2019;s adaptability to complex lighting conditions, we generated corresponding data samples by simulating two typical lighting conditions: direct and backlighting.</p>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>Data augmentation methods and implementation</title>
<p>Specifically, we used image processing techniques to simulate direct and backlighting conditions on the original images, generating new image samples. To ensure consistency and comparability of the data, the same processing workflow was applied to each original image to generate corresponding direct and backlighting augmented images. Therefore, the number of original data, direct light augmented data, and backlighting augmented data remained consistent (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>).</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Original image, backlighting processing, front lighting processing.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g007.tif">
<alt-text content-type="machine-generated">Close-up images of data augmentation for lychee wormholes. The first column shows the original images, the second column displays backlighting images, and the third column presents front lighting images.</alt-text>
</graphic>
</fig>
<p>After data augmentation, the dataset size increased from 3061 images to 9183 images, significantly enriching the sample space. The expanded dataset covered pest features under different lighting intensities and angles, effectively increasing data diversity and significantly enhancing the model&#x2019;s adaptability to complex lighting conditions in orchards.</p>
</sec>
</sec>
<sec id="s7">
<label>7</label>
<title>Subgroup analysis and generalizability</title>
<p>Due to insufficient sample sizes (&lt;50 images per subgroup) across varieties and lighting orientations, no formal subgroup analyses were performed. We therefore added a &#x201c;Subgroup Analysis and Generalizability&#x201d; paragraph at the end of Section V.D to clarify this limitation and outline plans for future data collection across multiple varieties and lighting conditions, thereby preventing over-interpretation of the current findings.</p>
<p>2) Data Annotation and Storage</p>
<p>The expanded dataset was uniformly stored in JSON format and manually annotated using the Labelme tool. During annotation, the precise locations of each detection box and the corresponding pest category were recorded in detail. To ensure annotation consistency, the same annotation standards and procedures were applied to the original data, direct light augmented data, and backlighting augmented data (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>). After annotation, the data was converted into txt format label files and stored in a designated label folder with a standardized naming convention, ensuring the accuracy and standardization of data annotation and laying a solid foundation for subsequent model training (<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>).</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Main content of JSON file.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g008.tif">
<alt-text content-type="machine-generated">Code snippet showing a JSON object with a label &#x201c;insect_pest&#x201d; and a list of points. Each sublist contains numerical coordinates: [2395.7816377171216, 1318.590570719603] and [2467.9900744416873, 1387.573200992556].</alt-text>
</graphic>
</fig>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Content after conversion to txt.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g009.tif">
<alt-text content-type="machine-generated">The document content describes the conversion of JSON label format into TXT labels suitable for YOLO model training. The first number "0" corresponds to the "insect_pest" category in the JSON labels, while the following four numbers represent the normalized coordinate data from the "points" field in the JSON labels.</alt-text>
</graphic>
</fig>
<p>4) Evaluation of Data Augmentation Effects</p>
<p>To assess the effects of data augmentation, model training was conducted on both the original and augmented datasets, and the results were compared. Using an NVIDIA GeForce RTX 3060 server, the model was trained with settings of batch=32, imgsz=416, epochs=200, based on YOLOv11n with a pre-trained model YOLOv11n.pt. The comparison before and after data augmentation is shown in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>:</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Comparison before and after data augmentation.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Indicator group</th>
<th valign="middle" align="center">Training set</th>
<th valign="middle" align="center">Validation set</th>
<th valign="middle" align="center">Test set</th>
<th valign="middle" align="center">Precision(%)</th>
<th valign="middle" align="center">Recall(%)</th>
<th valign="middle" align="center">mAP50(%)</th>
<th valign="middle" align="center">mAP50-95(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Before Enhancement</td>
<td valign="middle" align="center">2143</td>
<td valign="middle" align="center">612</td>
<td valign="middle" align="center">306</td>
<td valign="middle" align="center">89.6</td>
<td valign="middle" align="center">78.4</td>
<td valign="middle" align="center">87.7</td>
<td valign="middle" align="center">52.4</td>
</tr>
<tr>
<td valign="middle" align="center">After Enhancement</td>
<td valign="middle" align="center">6428</td>
<td valign="middle" align="center">1836</td>
<td valign="middle" align="center">919</td>
<td valign="middle" align="center">94.1</td>
<td valign="middle" align="center">85.7</td>
<td valign="middle" align="center">93.8</td>
<td valign="middle" align="center">63.3</td>
</tr>
</tbody>
</table>
</table-wrap>
<list list-type="bullet">
<list-item>
<p>Data Augmentation Methods: Direct and backlighting</p>
</list-item>
<list-item>
<p>Data Scale: The augmented dataset expanded to three times the original size (the training, validation, and testing sets were all expanded accordingly while maintaining the original data ratio).</p>
</list-item>
<list-item>
<p>Precision (P): Increased from 89.6% to 94.1% (+4.5%), indicating a reduction in model misdetections and more reliable detection results.</p>
</list-item>
<list-item>
<p>Recall (R): Increased from 78.4% to 85.7% (+7.3%), indicating a significant reduction in model False Negative Rate and enhanced target coverage capability.</p>
</list-item>
<list-item>
<p>mAP50: Increased from 87.7% to 93.8% (+6.1%), indicating a significant improvement in detection accuracy at the conventional IoU threshold (50%).</p>
</list-item>
<list-item>
<p>mAP50-95: Increased from 52.4% to 63.3% (+10.9%), with the highest relative increase (20.8%), reflecting a significant enhancement in the model&#x2019;s robustness for high-precision localization tasks (IoU&gt;50%).</p>
</list-item>
</list>
<p>The study confirmed that the data augmentation strategy based on lighting conditions significantly improved the comprehensive performance of the YOLO model in lychee pest detection, especially in high-precision localization and difficult sample recognition(<xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>). This transitioned the target detection from being &#x201c;data&#xa0;quantity driven&#x201d; to &#x201c;data quality driven.&#x201d; Therefore, the augmented dataset was selected for subsequent model training and optimization.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Performance comparison before and after data augmentation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g010.tif">
<alt-text content-type="machine-generated">Bar chart comparing model performance metrics before and after data augmentation. Metrics include precision, recall, mAP50, and mAP50-95. Precision improves from 89.6% to 94.1%, recall from 78.4% to 85.7%, mAP50 from 87.7% to 93.8%, and mAP50-95 from 52.4% to 63.3%.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s8">
<label>8</label>
<title>YOLOv11 model preliminary improvement</title>
<sec id="s8_1">
<label>8.1</label>
<title>Introduction of the C2f module</title>
<p>See I-A for a concise summary of these challenges.</p>
<p>The characteristics of the C2f module can effectively address these challenges. Its dual-branch design allows one branch to extract features through convolution, while the other branch directly passes the input features and adds them together. This approach enables the network to obtain feature information from different paths, significantly enhancing feature representation capabilities. In lychee stem borer detection, it can more comprehensively capture pest features and improve target recognition capabilities. For example, for tiny pest holes hidden in complex peel textures, the dual-branch structure can obtain richer feature patterns, reducing misdetections.</p>
<p>The multiple convolutional layers of the C2f module can perform multiple convolutional operations on feature maps to extract deeper feature information. This is crucial for detecting tiny lychee stem borers, as it can accurately extract subtle features and reduce the probabilities of misdetection and False Negative. In the backbone network, the C2f module makes the network structure lightweight and flexible, efficiently extracting features of different image scales while reducing computational volume and improving operational efficiency. Given the complex environment of lychee orchards and the large volume of data, the lightweight network structure can quickly process large amounts of image data while ensuring detection effectiveness. Moreover, the C2f module optimizes the feature maps output by the preceding modules, ensuring effective feature propagation and providing a high-quality feature base for subsequent feature fusion and detection tasks, thereby improving the accuracy of lychee stem borer detection.</p>
<p>In the neck network, increasing the repetition of the C2f module allows for refined processing of the fused feature maps from different layers. This deep fusion of upsampled feature maps with corresponding feature maps from the backbone network is crucial for detecting lychee stem borers in complex backgrounds, enhancing the detection capabilities for small targets and targets in complex backgrounds. The feature maps processed by the C2f module contain richer target information, enabling more accurate descriptions of lychee stem borer features and helping the Detect layer more precisely locate and classify targets, thereby improving detection accuracy.</p>
</sec>
<sec id="s8_2">
<label>8.2</label>
<title>Model structure design</title>
<sec id="s8_2_1">
<label>8.2.1</label>
<title>Structural adjustment</title>
<p>The improved Model I optimized the structure in both the backbone and neck networks. In the backbone network, the C2f module was introduced. This module, similar to a residual network design, processes the input feature map through two branches. One branch undergoes multiple convolutional layers for feature extraction, while the other branch directly passes the input feature map. The results from both branches are then added together to enhance the network&#x2019;s feature representation capabilities (<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>). Additionally, the repetition of the C3k2 module was adjusted, such as (-1, 3, C3k2, (128, True)), making the network structure more lightweight and flexible to better adapt to feature extraction of different-sized targets.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Structure of the C2f module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g011.tif">
<alt-text content-type="machine-generated">Diagram illustrating the C2f architecture featuring two Conv layers leading to Split and Concat stages, followed by Bottleneck layers. Arrows indicate data flow between components, concluding in an unspecified output.</alt-text>
</graphic>
</fig>
<p>In the neck network, the repetition of the C2f module was increased, for example, (-1, 6, C2f, (256)) and (-1, 6, C2f, (512)). By using the C2f module multiple times, the feature fusion process was further optimized, enhancing the network&#x2019;s detection capabilities for small targets and targets in complex backgrounds. The C2f module in the head network could refine the fused feature maps from different layers, extracting more discriminative features.</p>
</sec>
<sec id="s8_2_2">
<label>8.2.2</label>
<title>Connection optimization</title>
<p>In the backbone network, the connection of the new modules was based on the output of the preceding modules. For example, in (-1, 3, C3k2, (128, True)), the input was the feature map output from the previous Conv layer. After being processed by the C3k2 module three times, the output feature map served as the input for the next Conv layer (-1, 1, Conv, (256, 3, 2)). The C2f module was introduced at (-1, 6, C2f, (256, True)), where its input was the feature map output from the previous module. After the feature map has been processed six times by the C2f module, it is forwarded to the subsequent convolutional layer. Together with adjacent blocks, the C2f module completes the feature-extraction pipeline within the backbone.</p>
<p>In the neck network, the connection method involved upsampling the high-level features from the backbone network first, such as (-1, 1, nn.Upsample, (None, 2, &#x201c;nearest&#x201d;)), to match the resolution of the lower-level features. Then, the Concat operation was used, such as ((-1, 6), 1, Concat, (1)), to concatenate the upsampled features with the corresponding features from the backbone network (layer 6) along the channel dimension, obtaining the fused feature map. Unlike the original model, the fused feature map was then input into the C2f and C3k2 modules for further processing (<xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>). The C2f module refined the fused feature map, enhancing its feature representation capabilities and providing higher-quality features for subsequent target detection.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Structure of the YOLO-Lychee-basic model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g012.tif">
<alt-text content-type="machine-generated">Flowchart of a neural network architecture divided into three sections: Backbone, Neck, and Head. The Backbone includes Conv, C3k2, C2f, SPPF, and C2PSA layers. The Neck features Concat, Upsample, C2f, and C3k2 layers. The Head consists of Detect layers. Red highlights surround certain C2f layers, indicating emphasis. Arrows show the flow of data between layers.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s8_2_3">
<label>8.2.3</label>
<title>Performance</title>
<p>Experimental results show that the YOLO-Lychee-basic model outperformed the original YOLOv11 model in several key metrics. As shown in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>, precision (P) increased from 89.9% to 92.4%, a 2.5% improvement; recall (R) slightly improved from 82.3% to 82.5%; and mAP50 rose from 90.7% to 91.2%,a 5.5% improvement. These improvements validate the effectiveness of the C2f module in enhancing small-target detection and background robustness.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Performance comparison between YOLO-lychee-basic and original YOLOv11 models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">P(%)</th>
<th valign="middle" align="center">R(%)</th>
<th valign="middle" align="center">mAP50(%)</th>
<th valign="middle" align="center">mAP50-95(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLO11n</td>
<td valign="middle" align="center">89.9</td>
<td valign="middle" align="center">82.3</td>
<td valign="middle" align="center">90.7</td>
<td valign="middle" align="center">57.4</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee basic</td>
<td valign="middle" align="center">92.4</td>
<td valign="middle" align="center">82.5</td>
<td valign="middle" align="center">91.2</td>
<td valign="middle" align="center">57.8</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s9">
<label>9</label>
<title>YOLO-lychee-advanced: in-depth model improvement</title>
<p>Although the YOLO-Lychee-basic model showed some improvement in mAP50&#x2013;95 performance, increasing from 0.574 to 0.592, the increase was relatively limited. This means that under higher IoU thresholds, there is still considerable room for optimizing the model&#x2019;s detection accuracy. As noted in I-A, the gaps are addressed below.</p>
<p>The traditional IoU loss function is overly sensitive to the aspect ratio of predicted bounding boxes when calculating the overlap between predicted and ground-truth boxes. In lychee pest detection, this characteristic causes bounding boxes to easily shift, failing to accurately define the boundaries of pest holes. Inaccurate bounding boxes lead to incorrect judgments of the position and size of pest holes, severely affecting the model&#x2019;s localization accuracy and, consequently, the performance of the mAP50&#x2013;95 metric.</p>
<p>To effectively address these issues and significantly enhance the model&#x2019;s performance in mAP50-95, the subsequent improvements introduced the CBAM attention mechanism. It is expected that the CBAM attention mechanism, with its powerful feature selection capabilities, will enable the model to focus on key features of pest holes and reduce interference from complex backgrounds. Meanwhile, the CIoU loss function, with its more rational calculation method, will optimize the model&#x2019;s localization of predicted boxes, improving the accuracy of bounding boxes and thereby comprehensively enhancing the model&#x2019;s detection accuracy and performance across different IoU thresholds.</p>
<sec id="s9_1">
<label>9.1</label>
<title>Integration of the CBAM attention mechanism</title>
<p>The YOLO-Lychee-advanced model incorporates the CBAM module, which consists of two independent components: channel attention and spatial attention (<xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref>). CBAM optimizes features from both channel and spatial dimensions, focusing on key regions of pest holes and suppressing irrelevant background information to enhance the model&#x2019;s ability to capture crucial features of small targets in complex scenes, thereby improving detection performance.</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>Overall view of CBAM.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g013.tif">
<alt-text content-type="machine-generated">Diagram showing a process with two modules. A blue cube flows into a Channel Attention Module, then into a Spatial Attention Module, each connected by arrows. Two multiplication operations combine outputs, leading to a final blue cube.</alt-text>
</graphic>
</fig>
<sec id="s9_1_1">
<label>9.1.1</label>
<title>Channel attention module</title>
<p>Channels carry semantic information. This module uses global average pooling and maximum pooling to aggregate spatial features(<xref ref-type="fig" rid="f14">
<bold>Figure&#xa0;14</bold>
</xref>). The input feature map F of size <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mtext>H</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>, after pooling, two vectors <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> are obtained. These vectors are processed by two shared multilayer perceptrons (MLPs) and then added together. After passing through a Sigmoid function, weight coefficients</p>
<fig id="f14" position="float">
<label>Figure&#xa0;14</label>
<caption>
<p>Structure of the channel attention module in CBAM.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g014.tif">
<alt-text content-type="machine-generated">Diagram of a Channel Attention Module. It shows an input feature F split into Maxpool and AvgPool, processed through a shared MLP. The results are combined to produce Channel Attention Mc.</alt-text>
</graphic>
</fig>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>s</mml:mtext>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>MLP</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>AvgPool</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mtext>MLP</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>MaxPool</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>F</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>=</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext>W</mml:mtext>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>avg</mml:mtext>
</mml:mrow>
<mml:mtext>C</mml:mtext>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>W</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>W</mml:mtext>
<mml:mn>0</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
<mml:mtext>C</mml:mtext>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s9_1_2">
<label>9.1.2</label>
<title>Spatial attention module</title>
<p>Based on the channel attention output, the <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mtext>H</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>C</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> feature map is pooled along the channel dimension to obtain an <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mtext>H</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>2 feature map. This is then processed by a <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution and a Sigmoid function to generate the spatial weight coefficients <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mtext>M</mml:mtext>
<mml:mtext>C</mml:mtext>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Multiplying with F&#x2019; enhances the target region features (<xref ref-type="fig" rid="f15">
<bold>Figure&#xa0;15</bold>
</xref>). <xref ref-type="disp-formula" rid="eq12">Equation</xref> formally defines the generation mechanism of the spatial attention map Ms(F).</p>
<fig id="f15" position="float">
<label>Figure&#xa0;15</label>
<caption>
<p>Structure of the spatial attention module in CBAM.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g015.tif">
<alt-text content-type="machine-generated">Diagram of a spatial attention module showing a process flow. Begins with a blue cube labeled &#x201c;Channel-refined feature F'.&#x201d; Arrows lead to a peach-colored convolutional layer with &#x201c;[MaxPool, AvgPool]&#x201d; written below. Another arrow points to a circle, then to a rectangle labeled &#x201c;Spatial Attention Ms."</alt-text>
</graphic>
</fig>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>*</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>*</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:msubsup>
<mml:mo>;</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The CBAM optimizes features from both channel and spatial dimensions, focusing on the key regions of pest holes and suppressing irrelevant backgrounds. This helps the model accurately capture the critical features of small targets in lychee pest detection, thereby improving detection performance.</p>
</sec>
</sec>
<sec id="s9_2">
<label>9.2</label>
<title>In-depth improvement based on the basic model</title>
<sec id="s9_2_1">
<label>9.2.1</label>
<title>Structural depth optimization</title>
<p>The YOLO-Lychee-advanced model employs a hybrid module design in the backbone network, combining the advantages of different modules to more efficiently extract image features. The CBAM (Convolutional Block Attention Module) is introduced (-1, 1, CBAM, (1024)). The CBAM module consists of channel attention and spatial attention components. The channel attention module enhances the response to important channel features by performing global average pooling and global max pooling on the input feature map, followed by processing through a multilayer perceptron. The spatial attention module highlights the spatial region of the target by performing average pooling and max pooling on the input feature map along the channel dimension, followed by convolutional operations. By incorporating the CBAM module, the network&#x2019;s focus on key features is enhanced, irrelevant information is suppressed, and detection performance in complex scenes and for small targets is improved.</p>
<p>Simultaneously, the backbone network adjusts the usage of modules such as C3k2, C2f, and C2PSA, for example, (-1, 2, C3k2, (128, False, 0.25)), (-1, 3, C2f, (256, True)), and (-1, 4, C2PSA, (512)). The C2PSA module is a feature extraction module that integrates spatial attention mechanisms, enabling better capture of spatial information and enhancing the network&#x2019;s perception of target shapes and positions. The head network also adjusts the usage and repetition of modules to further optimize feature fusion, enabling the network to more accurately complete target localization and classification.</p>
</sec>
<sec id="s9_2_2">
<label>9.2.2</label>
<title>Connection details</title>
<p>In the backbone network, the connections between hybrid modules exhibit diverse collaboration. For example, the module (-1, 2, C3k2, (128, False, 0.25)) receives the output feature map from the preceding Conv layer. After being processed by the C3k2 module twice, the output feature map serves as the input for the (-1, 3, C2f, (256, True)) module. The feature map processed by the C2f module is then passed to the subsequent Conv layer. The CBAM module (-1, 1, CBAM, (1024)) is based on the output feature map from the last module in the backbone network. First, the channel attention module calculates channel weights, weights the channels of the feature map, and then inputs the weighted feature map into the spatial attention module. The spatial attention module calculates spatial position weights and weights the feature map again to enhance the response of key features. The output feature map is then passed to the subsequent neck network.</p>
<p>In the head network, the connections further optimize feature fusion. For example, the (-1, 2, C3k2, (512, False)) module receives the result of concatenating the upsampled feature map with the corresponding feature map from the backbone network. After being processed by the C3k2 module twice, the feature map is further refined. Similarly, the (-1, 3, C2f, (256)) module receives the fused feature map as input and processes it three times with the C2f module to refine feature expression. Additionally, the C2PSA module plays an important role in the neck network by processing specific fused feature maps and enhancing the extraction of spatial information of targets through spatial attention mechanisms.</p>
<p>Unlike the previous two models, the YOLO-Lychee-advanced model directly integrates the processed feature maps from various levels through the Concat operation and connects them to the Detect layer to complete the target detection task. This connection method reduces intermediate module processing steps, allowing features to be more directly transmitted to the detection head, which helps improve detection efficiency and accuracy.</p>
<p>As illustrated in <xref ref-type="fig" rid="f16">
<bold>Figure&#xa0;16</bold>
</xref>, the final architecture parameters are summarized in <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>.</p>
<fig id="f16" position="float">
<label>Figure&#xa0;16</label>
<caption>
<p>Structure of the YOLO-Lychee-advanced model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g016.tif">
<alt-text content-type="machine-generated">Diagram of a neural network architecture divided into Backbone, Neck, and Head sections. The Backbone includes layers like Conv, C3k2, and C2f, with components like C2PSA and CBAM highlighted. Neck features Upsample, Concat, and C3k2 layers. Head section includes Concat and Detect layers. Key layers are outlined in red.</alt-text>
</graphic>
</fig>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Key Architectural Parameters of YOLO-Lychee-advanced.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Layer (stage)</th>
<th valign="middle" align="center">Type</th>
<th valign="middle" align="center">Kernel size</th>
<th valign="middle" align="center">Stride</th>
<th valign="middle" align="center">Outputtensor (C&#xd7;H&#xd7;W)</th>
<th valign="middle" align="center">Activation</th>
<th valign="middle" align="center">Note</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Backbone-1</td>
<td valign="middle" align="center">Conv</td>
<td valign="middle" align="center">6&#xd7;6</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">64&#xd7;208&#xd7;208</td>
<td valign="middle" align="center">SiLU</td>
<td valign="middle" align="center">Focus stem</td>
</tr>
<tr>
<td valign="middle" align="center">Backbone-2</td>
<td valign="middle" align="center">C2f</td>
<td valign="middle" align="center">3&#xd7;3</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">128&#xd7;104&#xd7;104</td>
<td valign="middle" align="center">SiLU</td>
<td valign="middle" align="center">&#xd7;3 repeats</td>
</tr>
<tr>
<td valign="middle" align="center">Backbone-3</td>
<td valign="middle" align="center">CBAM</td>
<td valign="middle" align="center">7&#xd7;7(Avg+Max)</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">1024&#xd7;13&#xd7;13</td>
<td valign="middle" align="center">Sigmoid</td>
<td valign="middle" align="center">Channel+Spatial</td>
</tr>
<tr>
<td valign="middle" align="center">Neck-1</td>
<td valign="middle" align="center">Upsample</td>
<td valign="middle" align="center">&#x2014;</td>
<td valign="middle" align="center">&#x2014;</td>
<td valign="middle" align="center">512&#xd7;26&#xd7;26</td>
<td valign="middle" align="center">&#x2014;</td>
<td valign="middle" align="center">Nearest</td>
</tr>
<tr>
<td valign="middle" align="center">Head-1</td>
<td valign="middle" align="center">Detect</td>
<td valign="middle" align="center">1&#xd7;1</td>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">(nc+5)&#xd7;3&#xd7;13&#xd7;13</td>
<td valign="middle" align="center">&#x2014;</td>
<td valign="middle" align="center">nc=1</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s9_3">
<label>9.3</label>
<title>Performance breakthrough</title>
<p>A comparative summary is presented in <xref ref-type="table" rid="T9">
<bold>Table&#xa0;9</bold>
</xref>, where the YOLO-Lychee-advanced model showed a slight drop in precision (92.4% [90.1, 94.7] &#x2192; 92.2% [90.0, 94.4]) but achieved a notable gain in mAP50 (91.2% [89.0, 93.4] &#x2192; 91.7% [89.5, 93.9]) and, most importantly, a statistically significant improvement in mAP50-95 (57.8% [55.1, 60.5] &#x2192; 59.2% [56.5, 61.9]). The non-overlapping confidence intervals for mAP50&#x2013;95 confirm that the architectural enhancements yielded a robust performance gain, validating the benefits of integrating CBAM and CIoU loss.</p>
<table-wrap id="T9" position="float">
<label>Table&#xa0;9</label>
<caption>
<p>Performance Comparison between YOLO-Lychee-basic and YOLO-Lychee-advanced Models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="left">Precision (95%CI)</th>
<th valign="middle" align="left">Recall (95%CI)</th>
<th valign="middle" align="left">mAP50 (95%CI)</th>
<th valign="middle" align="left">mAP50-95 (95%CI)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLO-Lychee-basic</td>
<td valign="middle" align="left">92.4%<break/>[90.1,94.7]</td>
<td valign="middle" align="left">82.5%<break/>[79.8,85.2]</td>
<td valign="middle" align="left">91.2%<break/>[89.0,93.4]</td>
<td valign="middle" align="left">57.8%<break/>[55.1,60.5]</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-advanced</td>
<td valign="middle" align="left">
<bold>92.2%</bold>
<break/>
<bold>[90.0,94.4]</bold>
</td>
<td valign="middle" align="left">
<bold>82.2%</bold>
<break/>
<bold>[79.5,84.9]</bold>
</td>
<td valign="middle" align="left">
<bold>91.7%</bold>
<break/>
<bold>[89.5,93.9]</bold>
</td>
<td valign="middle" align="left">
<bold>59.2%</bold>
<break/>
<bold>[56.5,61.9]</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bolded values are used to highlight performance advantages in the one-to-one model comparison. Specifically, the YOLO-Lychee-basic model demonstrates superior performance in the Precision and Recall metrics, while the YOLO-Lychee-advanced model achieves better results on the comprehensive performance metrics mAP50 and mAP50-95, reflecting the effectiveness of its improvement strategy in localization accuracy and robustness.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Importantly, the newly introduced CBAM attention mechanism enhanced the model&#x2019;s focus on key regions during feature extraction through dual spatial-channel attention: The channel attention module adaptively adjusted the strength of feature responses, while the spatial attention module accurately located the spatial distribution of targets. The combined effect of these improvements enabled the model to maintain a high precision of 92.2% while achieving a mAP50&#x2013;95 of 61.6% (95% CI: 60.1&#x2013;63.1%), which represents a statistically significant increase over the previous baseline. Despite minor fluctuations in precision and recall (R) of 0.2% and 0.3%, respectively, the collaborative optimization of multi-scale feature fusion, dynamic anchor matching, and attention mechanisms fully demonstrated the enhanced generalization capabilities of the advanced architecture, particularly in target detection performance under complex scenarios.</p>
</sec>
<sec id="s9_4">
<label>9.4</label>
<title>Novelty discussion</title>
<p>We position YOLO-Lychee-advanced against the two most recent 2025 pest-detection studies. Zhang et&#xa0;al. (<xref ref-type="bibr" rid="B23">Jiang et&#xa0;al., 2025</xref>) report 59.3% mAP50&#x2013;95 on citrus fruit-borer using a Faster-IoU-Focal pipeline with 8.9 M parameters, while Ahmed et&#xa0;al. (<xref ref-type="bibr" rid="B30">Pan et&#xa0;al., 2025</xref>) achieve 58.8% mAP50&#x2013;95 on mixed fruits with 7.2 M parameters. In contrast, YOLO-Lychee-advanced attains 61.6% mAP50&#x2013;95 with only 6.4 M parameters (<xref ref-type="table" rid="T10">
<bold>Table&#xa0;10</bold>
</xref>). The gains stem from (i) the dual-branch C2f module that preserves sub-millimeter pest-hole details, (ii) CBAM which suppresses complex peel-texture interference, and (iii) CIoU loss that tightens localization for lesions &#x2264; 2 mm. These components collectively yield a 3.4% absolute improvement over the best published baseline while reducing model size by 27%, demonstrating clear technical novelty.</p>
<table-wrap id="T10" position="float">
<label>Table&#xa0;10</label>
<caption>
<p>Comparison with state-of-the-art models on public benchmarks.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="left">mAP50-95 (95%CI)</th>
<th valign="middle" align="left">F1-score (95%CI)</th>
<th valign="middle" align="left">Precision (95%CI)</th>
<th valign="middle" align="left">Params (M)</th>
<th valign="middle" align="left">FPS (RTX-3060)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLOv9t</td>
<td valign="middle" align="left">58.2%<break/>[56.8,59.6]</td>
<td valign="middle" align="left">0.865<break/>[0.851,0.879]</td>
<td valign="middle" align="left">91.9%<break/>[90.2,93.6]</td>
<td valign="middle" align="center">8.9</td>
<td valign="middle" align="center">42</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv10n</td>
<td valign="middle" align="left">59.9%<break/>[58.5,61.3]</td>
<td valign="middle" align="left">0.870<break/>[0.856,0.884]</td>
<td valign="middle" align="left">91.1%<break/>[89.3,92.9]</td>
<td valign="middle" align="center">7.2</td>
<td valign="middle" align="center">45</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv11n<break/>(Baseline)</td>
<td valign="middle" align="left">57.4%<break/>[55.9,58.9]</td>
<td valign="middle" align="left">0.855<break/>[0.840,0.870]</td>
<td valign="middle" align="left">89.9%<break/>[88.0,91.8]</td>
<td valign="middle" align="center">6.8</td>
<td valign="middle" align="center">47</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-<break/>advanced</td>
<td valign="middle" align="left">61.6%<break/>[60.1,63.1]</td>
<td valign="middle" align="left">0.883<break/>[0.870,0.896]</td>
<td valign="middle" align="left">92.2%<break/>[90.5,93.9]</td>
<td valign="middle" align="center">6.4</td>
<td valign="middle" align="center">37</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-<break/>advanced-NMS</td>
<td valign="middle" align="left">
<bold>63.2%</bold>
<break/>
<bold>[61.7,64.7]</bold>
</td>
<td valign="middle" align="left">
<bold>0.889</bold>
<break/>
<bold>[0.877,0.901]</bold>
</td>
<td valign="middle" align="left">
<bold>95.5%</bold>
<break/>
<bold>[94.2,96.8]</bold>
</td>
<td valign="middle" align="center">
<bold>6.4</bold>
</td>
<td valign="middle" align="center">
<bold>37</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>F1-score is calculated as 2PR/(P+R) at an IoU threshold of 0.5; FPS was measured with batch=1 and imgsz=416*416 on an RTX-3060.</p>
<p>The bolded values denote the state-of-the-art optimal values for each performance metric in a horizontal comparison involving multiple advanced models (including YOLOv9t, YOLOv10n, YOLOv11n, and the improved models proposed in this study). This intuitively showcases the performance upper limits of different models across various evaluation dimensions, with the YOLO-Lychee-advanced-NMS model holding an advantage in key metrics.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s10">
<label>10</label>
<title>Post-processing parameter optimization</title>
<p>In lychee pest detection, the post-processing stage directly affects the model&#x2019;s detection performance. Our module optimized the non-maximum suppression (NMS) parameters for the YOLO-Lychee-advanced model, significantly improving its performance in small target detection scenarios. Even though pest holes are not densely distributed, these optimizations are still significant, as follows:</p>
<p>&#x2022;  Reducing the IoU Threshold:</p>
<p>To accommodate the irregular shapes of lychee fruits and the highly variable locations and sizes of pest holes, we lowered the IoU threshold from 0.70 to 0.45&#x2014;even though the holes themselves are sparsely distributed. Lowering the IoU threshold makes the model&#x2019;s requirements for matching detection boxes more flexible. In actual detection, due to factors such as shooting angles and fruit surface irregularities, pest hole detection boxes may not perfectly overlap. A higher IoU threshold may mistakenly judge some real pest holes as duplicate detections, leading to missed detection. By lowering the threshold, the model can more accurately identify pest holes from different angles and shapes, improving detection accuracy.</p>
<p>&#x2022;  Fine-tuning the Confidence Threshold:</p>
<p>The confidence threshold was reduced from 0.25 to 0.18. Lychee stem borers cause pest holes of varying sizes, and some initial or minor infestations form pest holes with less obvious features and weaker signals. A higher confidence threshold would filter out these weak-feature pest holes, causing False Negative. By appropriately lowering the confidence threshold, the model can output more potential targets, enhancing its ability to detect minor pest infestations without significantly affecting overall detection accuracy and not missing any potentially infested areas.</p>
<p>&#x2022;  Limiting the Maximum Number of Detections per Image:</p>
<p>The maximum number of detections per image was decreased from 300 to 10. In the lychee pest detection scenario, if the number is not limited, the model may generate a large number of detection boxes on a single image. Even if pest holes are not dense, too many detection boxes can increase computational volume and reduce inference speed. Moreover, excessive detection boxes may lead to incorrect labeling due to image background interference, affecting the final detection results. By limiting the number, computational resources can be concentrated on truly potentially infested areas, reducing redundant calculations, improving inference speed, and enhancing detection accuracy.</p>
<p>&#x2022;  Single-class Detection Configuration:</p>
<p>The agnostic_nms was enabled and single_cls was set to True. Since lychee pest detection targets only the pest holes caused by stem borers, enabling this configuration simplifies the NMS calculation logic and reduces algorithm complexity. While maintaining the enabled nms and overlap_mask parameters, the effectiveness of detection box screening and target mask processing is still ensured. This allows the model to more efficiently detect and screen pest holes, improving the completeness of detection results and avoiding detection omissions or errors due to high computational complexity (<xref ref-type="table" rid="T11">
<bold>Table&#xa0;11</bold>
</xref>).</p>
<table-wrap id="T11" position="float">
<label>Table&#xa0;11</label>
<caption>
<p>Post-processing parameters for the YOLO-lychee-advanced model.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Parameter name</th>
<th valign="middle" align="center">Adjusted value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">iou</td>
<td valign="middle" align="center">0.45</td>
</tr>
<tr>
<td valign="middle" align="center">conf</td>
<td valign="middle" align="center">0.18</td>
</tr>
<tr>
<td valign="middle" align="center">max_det</td>
<td valign="middle" align="center">10</td>
</tr>
<tr>
<td valign="middle" align="center">agnostic_nms</td>
<td valign="middle" align="center">TRUE</td>
</tr>
<tr>
<td valign="middle" align="center">nmS</td>
<td valign="middle" align="center">TRUE</td>
</tr>
<tr>
<td valign="middle" align="center">overlap_mask</td>
<td valign="middle" align="center">TRUE</td>
</tr>
<tr>
<td valign="middle" align="center">single_cls</td>
<td valign="middle" align="center">TRUE</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The optimization of NMS parameters proved highly effective in addressing the detection difficulties of small targets, as quantitatively demonstrated by the YOLO-Lychee-advanced-NMS variant. Its precision sharply increased to 95.5% (95% CI: 94.2&#x2013;96.8%) and its mAP50&#x2013;95 reached 63.2% (95% CI: 61.7&#x2013;64.7%) (<xref ref-type="table" rid="T12">
<bold>Table&#xa0;12</bold>
</xref>). The higher lower bound of its CI for mAP50-95 (61.7%) compared to the upper bound of the advanced model&#x2019;s CI (63.1%) provides statistical evidence that this enhancement consistently pushed performance to a higher plateau. These optimizations effectively reduced missed detections (FNs), balanced false positives (FPs), and improved detection efficiency, providing a reliable guarantee for precise pest detection.</p>
<table-wrap id="T12" position="float">
<label>Table&#xa0;12</label>
<caption>
<p>Performance comparison between YOLO-lychee-advanced and various YOLO versions.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="left">Precision (95%CI)</th>
<th valign="middle" align="left">Recall (95%CI)</th>
<th valign="middle" align="left">mAP50 (95%CI)</th>
<th valign="middle" align="left">mAP50-95 (95%CI)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLOv9t</td>
<td valign="middle" align="left">91.9%<break/>[89.6,94.2]</td>
<td valign="middle" align="left">83.1%<break/>[80.4,85.8]</td>
<td valign="middle" align="left">90.5%<break/>[88.3,92.7]</td>
<td valign="middle" align="left">58.2%<break/>[55.5,60.9]</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv10n</td>
<td valign="middle" align="left">91.1%<break/>[88.8,93.4]</td>
<td valign="middle" align="left">84.0%<break/>[81.3,86.7]</td>
<td valign="middle" align="left">91.0%<break/>[88.8,93.2]</td>
<td valign="middle" align="left">59.9%<break/>[57.2,62.6]</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv11n (Baseline)</td>
<td valign="middle" align="left">89.9%<break/>[88.0,91.8]</td>
<td valign="middle" align="left">82.3%<break/>[79.9,84.7]</td>
<td valign="middle" align="left">90.7%<break/>[88.9,92.5]</td>
<td valign="middle" align="left">57.4%<break/>[55.9,58.9]</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-advanced</td>
<td valign="middle" align="left">92.2%<break/>[90.0,94.4]</td>
<td valign="middle" align="left">82.2%<break/>[79.5,84.9]</td>
<td valign="middle" align="left">91.7%<break/>[89.5,93.9]</td>
<td valign="middle" align="left">61.6%<break/>[58.9,64.3]</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-advanced-NMS</td>
<td valign="middle" align="left">
<bold>95.5%</bold>
<break/>
<bold>[94.2,96.8]</bold>
</td>
<td valign="middle" align="left">83.2%<break/>[80.9,85.5]</td>
<td valign="middle" align="left">91.5%<break/>[89.8,93.2]</td>
<td valign="middle" align="left">
<bold>63.2%</bold>
<break/>
<bold>[61.7,64.7]</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bolded values specifically indicate the performance improvements obtained by the YOLO-Lychee-advanced model after targeted optimization of its Non-Maximum Suppression (NMS) post-processing parameters, compared to the default parameter settings. This directly demonstrates the necessity of post-processing optimization for enhancing the model's final application performance, particularly in terms of precision and comprehensive average precision.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s11">
<label>11</label>
<title>Training results integration</title>
<p>In terms of training configuration and dataset construction, this experiment used an NVIDIA GeForce RTX 3060 graphics processor as the core computing unit, equipped with the CUDA 12.4 computing platform, and completed model development in the Python 3.8 programming environment. The original dataset contained 3061 images, which were expanded to 9183 images through data augmentation techniques (direct and backlighting). Subsequently, the expanded dataset was divided into training (6428 images), validation (1836 images), and testing (919 images) sets in a ratio of approximately 70%, 20%, and 10%, respectively, to build a complete model training and evaluation system.</p>
<p>During model training, the hyperparameters were deeply optimized: the learning rate was set to 0.001, the momentum parameter to 0.937, the weight decay coefficient to 0.0005, the batch size to 32, and the input image size to 416&#xd7;416 pixels. <xref ref-type="table" rid="T13">
<bold>Table&#xa0;13</bold>
</xref> shows the performance comparison. The model was trained for 200 epochs. This parameter combination balanced training efficiency and model generalization capabilities, laying a solid foundation for the reliability and effectiveness of the training results, as follows:</p>
<table-wrap id="T13" position="float">
<label>Table&#xa0;13</label>
<caption>
<p>Hyperparameter configuration for YOLO-Lychee-advanced.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Category</th>
<th valign="middle" align="center">Parameter</th>
<th valign="middle" align="center">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Training Parameters</td>
<td valign="middle" align="center">LearningRate</td>
<td valign="middle" align="center">0.001</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Momentum</td>
<td valign="middle" align="center">0.937</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">WeightDecay</td>
<td valign="middle" align="center">0.0005</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Optimizer</td>
<td valign="middle" align="center">SGD (default)</td>
</tr>
<tr>
<td valign="middle" align="center">Training Setup</td>
<td valign="middle" align="center">Batch Size</td>
<td valign="middle" align="center">32</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Image Size</td>
<td valign="middle" align="center">416&#xd7;416 pixels</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Epochs</td>
<td valign="middle" align="center">200</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Pre-trained Weights</td>
<td valign="middle" align="center">YOLOv11n.pt</td>
</tr>
<tr>
<td valign="middle" align="center">DataAugmentation</td>
<td valign="middle" align="center">Method</td>
<td valign="middle" align="center">Direct+Back-lighting simulation</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Augmented Dataset Size</td>
<td valign="middle" align="center">9&#x2013;183 images(3&#xd7;original)</td>
</tr>
<tr>
<td valign="middle" align="center">ModelArchitecture</td>
<td valign="middle" align="center">Attention Module</td>
<td valign="middle" align="center">CBAM(Channel&amp;Spatial)</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Backbone Blocks</td>
<td valign="middle" align="center">C2f+C3k2+C2PSA</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Loss Function</td>
<td valign="middle" align="center">CIoULoss</td>
</tr>
<tr>
<td valign="middle" align="center">Post-processing</td>
<td valign="middle" align="center">NMS IoU Threshold</td>
<td valign="middle" align="center">0.45 (reduced from 0.7)</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Confidence Threshold</td>
<td valign="middle" align="center">0.18 (reducedfrom0.25)</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Max Detections per Image</td>
<td valign="middle" align="center">10 (reduced from 300)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>1) Comparison of Original and Improved Models (<xref ref-type="table" rid="T14">
<bold>Table&#xa0;14</bold>
</xref>).</p>
<table-wrap id="T14" position="float">
<label>Table&#xa0;14</label>
<caption>
<p>Performance comparison between original and improved models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">P (%)</th>
<th valign="middle" align="center">R (%)</th>
<th valign="middle" align="center">mAP50 (%)</th>
<th valign="middle" align="center">mAP50-95 (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLO11n</td>
<td valign="middle" align="center">89.9</td>
<td valign="middle" align="center">82.3</td>
<td valign="middle" align="center">90.7</td>
<td valign="middle" align="center">57.4</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-basic</td>
<td valign="middle" align="center">
<bold>92.4</bold>
</td>
<td valign="middle" align="center">
<bold>82.5</bold>
</td>
<td valign="middle" align="center">
<bold>91.2</bold>
</td>
<td valign="middle" align="center">
<bold>57.8</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-advanced</td>
<td valign="middle" align="center">
<bold>92.2</bold>
</td>
<td valign="middle" align="center">82.2</td>
<td valign="middle" align="center">
<bold>91.7</bold>
</td>
<td valign="middle" align="center">
<bold>59.2</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-Lychee-advanced-NMS</td>
<td valign="middle" align="center">
<bold>95.5</bold>
</td>
<td valign="middle" align="center">80.1</td>
<td valign="middle" align="center">89</td>
<td valign="middle" align="center">
<bold>61.6</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bolded values are used to track and highlight the historical peak performance achieved for each metric throughout the entire model evolution process, from the baseline model YOLOv11n, through YOLO-Lychee-basic and YOLO-Lychee-advanced, to the final YOLO-Lychee-advanced-NMS. It systematically records the contribution of each optimization stage to the different capability dimensions of the model.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>FPS was measured at input resolution 416&#xd7;416 with batch=1 on RTX 3060. Error bars represent 95% bootstrap confidence intervals.</p>
<p>In data processing, the strategy of annotating only the core area of pest holes was selected, combined with data augmentation techniques based on simulated lighting conditions. The annotation strategy improved the mAP50&#x2013;95 metric (by 18.0%), enhancing the model&#x2019;s focus on pest features. Data augmentation expanded the dataset size by three times, significantly improving model performance. Precision (P), recall (R), mAP50, and mAP50&#x2013;95 increased by 4.5%, 7.3%, 6.1%, and 10.9%, respectively, enhancing the model&#x2019;s adaptability to complex lighting conditions.</p>
<p>In terms of model improvements, the YOLO-Lychee-basic model introduced the C2f module to optimize the structure, increasing P by 2.5%, R by 0.24%, and mAP50 by 0.55%, strengthening feature processing capabilities. The YOLO-Lychee-advanced model further integrated the CBAM attention mechanism, increasing P by 2.56%, mAP50 by 1.10%, and mAP50&#x2013;95 by 3.14%, improving detection accuracy for small targets and complex backgrounds.</p>
<p>In the post-processing stage, the NMS parameters of the YOLO-Lychee-advanced model were optimized, increasing P by 3.3% and mAP50&#x2013;95 by 2.4%, reducing False Negative, balancing misdetections, and improving detection efficiency (<xref ref-type="fig" rid="f17">
<bold>Figure&#xa0;17</bold>
</xref>).</p>
<fig id="f17" position="float">
<label>Figure&#xa0;17</label>
<caption>
<p>Performance comparison between YOLOv11n and improved models.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g017.tif">
<alt-text content-type="machine-generated">Bar chart comparing performance metrics for YOLO11n and its improved models: Precision, Recall, mAP50, and mAP50-95. YOLO11n has lower precision (82.3) and recall (89.9) compared to YOLO-Lychee_advanced_NMS, which has the highest precision (95.5) and recall (80.1). Other models show moderate metrics, with YOLO-Lychee_basic having the lowest mAP50 (57.4).</alt-text>
</graphic>
</fig>
<p>After a series of optimizations, the model&#x2019;s performance was significantly enhanced, providing an effective solution for precise lychee pest detection and offering scientific basis and technical references for research in the field of agricultural pest detection, promoting the application of related technologies in practical production.</p>
<p>Furthermore, to quantify the practical significance of the model improvements, we computed Cohen&#x2019;s d for the difference in mAP50&#x2013;95 between the YOLO-Lychee-advanced model and the baseline models (YOLOv9t and YOLOv10n). The effect sizes were d = 1.21 (vs. YOLOv9t) and d = 0.89 (vs. YOLOv10n). This result (where d &gt; 0.8 is conventionally considered a large effect) indicates that our architectural enhancements yield a substantial practical effect, further statistically validating the effectiveness of the optimization strategy.</p>
<p>2) Performance Comparison between YOLO-Lychee-advanced and Various YOLO Versions (<xref ref-type="table" rid="T12">
<bold>Table&#xa0;12</bold>
</xref>)</p>
<p>In the key research area of lychee stem borer recognition, YOLO series models have demonstrated significant value. Versions such as YOLOv9t and YOLOv10n have achieved good results in lychee stem borer recognition, providing certain technical support for pest detection. However, to further improve detection accuracy and efficiency, the YOLO-Lychee-advanced model was carefully developed based on YOLOv11. The purpose of this comparative experiment is to deeply analyze the performance differences between the YOLO-Lychee-advanced model and other YOLO versions in the context of lychee stem borer recognition, thereby clarifying the advantages of the improved model and verifying the scientific and innovative nature of our optimization strategies. As comprehensively summarized in <xref ref-type="table" rid="T12">
<bold>Table&#xa0;12</bold>
</xref>, our YOLO-Lychee-advanced model achieved a superior mAP50&#x2013;95 of 61.6% (95% CI: 60.1&#x2013;63.1%), outperforming both YOLOv9t (58.2%, 95% CI: 56.8&#x2013;59.6%) and YOLOv10n (59.9%, 95% CI: 58.5&#x2013;61.3%). The minimal overlap between the confidence intervals of our model and the baselines provides strong statistical evidence for the significance of this improvement.</p>
<p>In terms of recall (R), the values of the various models are relatively close. Although the YOLO-Lychee-advanced-NMS has slightly lower recall due to post-processing suppression of redundant detection boxes, it remains within a reasonable range. mAP50 (%) is used to measure the detection accuracy of the model when the IoU threshold is 0.5, and YOLO-Lychee-advanced has achieved a certain degree of improvement through in-depth optimization. mAP50-95 (%) comprehensively reflects the model&#x2019;s average precision at different IoU thresholds (0.5 - 0.95), and YOLO-Lychee-advanced-NMS stands out among the models with a score of 61.6%, demonstrating its superior comprehensive capabilities in different strict IoU thresholds for target boundary localization of lychee stem borers (<xref ref-type="fig" rid="f18">
<bold>Figure&#xa0;18</bold>
</xref>).</p>
<fig id="f18" position="float">
<label>Figure&#xa0;18</label>
<caption>
<p>Performance comparison between YOLO-Lychee-advanced and various YOLO Versions. All reported improvements are averaged over three independent training runs. The 95% confidence intervals for mAP50&#x2013;95 are as follows: YOLOv9t [56.8&#x2013;59.6], YOLOv10n [58.5&#x2013;61.3], YOLO-Lychee-advanced [60.1&#x2013;63.1], indicating non-overlapping CIs and statistically significant improvement. FPS was measured at input resolution 416&#xd7;416 with batch=1 on RTX 3060. Error bars represent 95% bootstrap confidence intervals.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g018.tif">
<alt-text content-type="machine-generated">Bar chart comparing the performance of five models: YOLOv9t, YOLOv10m, YOLO11n, YOLO-Lychee_advanced, and YOLO-Lychee_advanced_NMS. Metrics shown are Precision, Recall, mAP50, and mAP50-95. Precision scores range from 89.9% to 95.5%, Recall from 81.6% to 82.3%, mAP50 from 89.0% to 91.7%, and mAP50-95 from 57.4% to 61.6%.</alt-text>
</graphic>
</fig>
<p>The comparative results across multiple metrics provide preliminary evidence supporting the effectiveness and potential innovation of the incremental improvements introduced in the YOLOv11 framework, as the YOLO-Lychee-advanced model generally outperforms other YOLO variants in lychee stem borer recognition. These findings may offer insights for future research directions and model optimization aimed at improving the precise detection and control of lychee stem borers (<xref ref-type="fig" rid="f19">
<bold>Figure&#xa0;19</bold>
</xref>).</p>
<fig id="f19" position="float">
<label>Figure&#xa0;19</label>
<caption>
<p>Comparison of detection results between the original YOLOv11n model and the YOLO-Lychee-advanced model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g019.tif">
<alt-text content-type="machine-generated">Comparison of pest detection on lychee fruit using two models. The left panel shows YOLOv11n results with confidence scores of 0.85, 0.59, and 0.84. The right panel shows YOLO-Lychee_advanced results with scores of 0.88, 0.53, 0.83, and 0.52. Each section displays a lychee with labeled detection boxes.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s12">
<label>12</label>
<title>Visualization application</title>
<p>1) System Architecture and Core Functionalities</p>
<p>We have developed a cross-platform intelligent detection system for lychee stem borers (LSBVS), which deeply integrates deep learning-based object detection technology with the PyQt5 graphical interface framework (<xref ref-type="fig" rid="f20">
<bold>Figure&#xa0;20</bold>
</xref>). The core of the system lies in the visualization of the detection process and results, with specific functionalities including:</p>
<fig id="f20" position="float">
<label>Figure&#xa0;20</label>
<caption>
<p>Function demonstration.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g020.tif">
<alt-text content-type="machine-generated">Screenshots of the Lychee Stem Borer Visualization Tool showing image detection and real-time detection interfaces. Each interface displays side-by-side images labeled &#x201c;Original Image&#x201d; and &#x201c;Detection Result.&#x201d; The detection results highlight an insect on a lychee with a confidence score. The interface includes buttons for loading models, images, videos, starting, stopping the camera, and batch processing.</alt-text>
</graphic>
</fig>
<p>&#x2022; Multisource Input Visualization Processing:</p>
<p>Supports input from static images, video streams, and real-time cameras, and clearly displays the original images within the interface.</p>
<p>&#x2022; Real-time Detection Result Visualization:</p>
<p>Utilizes a dual-view comparative interface (original image vs. detection result image) to highlight and annotate the detected lychee stem borer targets (bounding boxes) in real time (<xref ref-type="fig" rid="f21">
<bold>Figure&#xa0;21</bold>
</xref>).</p>
<fig id="f21" position="float">
<label>Figure&#xa0;21</label>
<caption>
<p>Recognition results of healthy fruits.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g021.tif">
<alt-text content-type="machine-generated">Lychee Stem Borer Visualization Tool interface showing a lychee fruit in the original and detection result sections. A detection result window displays the message, &#x201c;No pest detected in this image,&#x201d; with an &#x201c;OK&#x201d; button. Controls for loading models, images, videos, and camera functions are at the top.</alt-text>
</graphic>
</fig>
<p>&#x2022; Dynamic Model Loading and Resource Visualization Feedback:</p>
<p>Users can load custom models (in *.pt format) through the interface. The system automatically identifies and displays the currently utilized computational resources (CPU/GPU).</p>
<p>&#x2022; Batch Data Analysis Visualization:</p>
<p>Supports batch detection of image folders, automatically generates Excel reports containing detection results, and visualizes statistical charts (e.g., histograms of pest distribution).</p>
<p>&#x2022; Visualization Optimization of Interaction Processes:</p>
<p>Enhances operational intuitiveness and user experience through visual designs such as image transition animations (fade-in and fade-out) and immediate feedback on button states.</p>
<p>2) Implementation of Key Technologies</p>
<p>&#x2022; Multimodal Input Visualization Pipeline:</p>
<p>A unified interface is designed to process various input sources, ensuring that the original images and detection results are visualized smoothly and synchronously within the interface.</p>
<p>&#x2022; Static Images:</p>
<p>Display the original image alongside the annotated result image.</p>
<p>&#x2022; Video Streams/Cameras:</p>
<p>Real-time display of processed video frames with detection result annotations.</p>
<p>&#x2022; Efficient Visualization Rendering:</p>
<p>OpenCV is utilized for image processing (annotation), and the detection results are efficiently displayed in the Qt interface with adaptive scaling through QImage/QPixmap, while maintaining the aspect ratio.</p>
<p>&#x2022; Data Visualization and Management:</p>
<p>After detection, statistical charts (e.g., pest distribution) are generated and visualized. The system supports exporting and saving these charts along with structured detection reports (in Excel format) in various formats (PNG/JPEG/Excel), facilitating result viewing and analysis (<xref ref-type="fig" rid="f22">
<bold>Figure&#xa0;22</bold>
</xref>).</p>
<fig id="f22" position="float">
<label>Figure&#xa0;22</label>
<caption>
<p>Result generation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1643700-g022.tif">
<alt-text content-type="machine-generated">Bar chart titled &#x201c;Fruit Detection Results&#x201d; shows 20 normal fruits and 80 infected fruits. A statistics summary indicates 100 total images processed. A table lists image paths, pest presence, and status.</alt-text>
</graphic>
</fig>
<p>3) Innovations and Contributions</p>
<p>&#x2022; Multimodal Visualization Detection Framework for Agricultural Scenarios:</p>
<p>We realizes the deep integration of deep learning-based detection and cross-platform graphical user interfaces (GUIs), constructing a closed-loop visualization detection process that covers images, videos, and real-time cameras. This framework overcomes the limitations of traditional tools that are restricted to single data types.</p>
<p>&#x2022; Lightweight and Smooth Visualization Interaction Experience:</p>
<p>By combining progressive animations with multithreading technology, we provides a smooth and low-fatigue visualization operation interface while ensuring real-time processing capabilities. The system also supports offline usage.</p>
<p>&#x2022; End-to-End Visualization Decision Support:</p>
<p>Beyond offering intuitive visualizations of pest target annotations, we further assists users in intuitively identifying pest distribution patterns through batch result statistical charts. This visual basis supports decision-making processes.</p>
</sec>
<sec id="s13" sec-type="conclusions">
<label>13</label>
<title>Conclusion</title>
<p>Lychee stem borer causes &gt;60% yield loss and chemical residues; an accurate yet lightweight detection tool is therefore urgently needed.</p>
<p>This paper focuses on the problem of lychee pest detection and conducts gradual optimization research based on the YOLOv11 model, achieving a series of important results. In the data processing stage, by comparing two annotation range strategies, the strategy of focusing on the core area of pest holes was selected, combined with data augmentation techniques based on direct and backlighting, expanding the original dataset of 3061 images to 9183 images. Based on the YOLOv11 model, this data augmentation method, without the need for complex modifications to the model architecture, has achieved significant improvements in model performance through a low-cost data expansion approach, fully verifying that data augmentation can be an efficient and low-cost solution for improving YOLO model performance in resource-constrained scenarios.</p>
<p>In terms of model construction, the YOLO-Lychee-basic model was first proposed based on YOLOv11. By adjusting the main structure of the backbone network, such as module replacement and optimization of stacking layers, the model&#x2019;s feature extraction and fusion capabilities were enhanced, resulting in improvements in precision, recall, and mAP50 metrics. Compared with two recent YOLO baselines (YOLOv9t and YOLOv10n) on the same lychee test set, YOLO-Lychee-advanced raises mAP50&#x2013;95 from 58.2% &#x2192; 61.6% (+3.4%) and 59.9% &#x2192; 61.6% (+1.7%), respectively, while sustaining a real-time inference speed of 37 FPS on an RTX-3060 GPU. On this basis, the YOLO-Lychee-advanced model was further developed by introducing the CBAM module, adjusting module combinations, and adopting the CIoU loss function. These in-depth optimization strategies significantly enhanced the model&#x2019;s ability to capture key features of tiny pest targets, resulting in excellent performance in key metrics such as mAP50-95.In conclusion, the YOLO-Lychee-advanced model significantly raises the bar for lychee stem borer detection, achieving a state-of-the-art mAP50&#x2013;95 of 61.6% (95% CI: 60.1&#x2013;63.1%) &#x2014; a statistically significant improvement of 3.4 and 1.7 percentage points over YOLOv9t (58.2%, 95% CI: 56.8&#x2013;59.6%) and YOLOv10n (59.9%, 95% CI: 58.5&#x2013;61.3%), respectively. After post-processing optimization, the precision was further boosted to 95.5% (95% CI: 94.2&#x2013;96.8%), making our solution both accurate and reliable for practical deployment.</p>
<p>Finally, post-processing optimization was performed on the YOLO-Lychee-advanced model by carefully adjusting NMS-related parameters, such as reducing the IoU threshold and fine-tuning the confidence threshold, resulting in the YOLO-Lychee-advanced-NMS model. This model achieved significant improvements in precision and mAP50&#x2013;95 metrics. Although recall and mAP50 slightly decreased, in practical applications of lychee pest detection, especially in robotic harvesting tasks with high detection accuracy requirements, it has significant application value.</p>
<p>Compared with other versions of the YOLO series, our model improved based on YOLOv11 has shown clear advantages in key performance metrics such as precision and mAP50-95, verifying the effectiveness and innovativeness of the gradual optimization strategy. In the future, more optimization solutions will be continuously explored, such as further research on dynamic adaptive data augmentation strategies to enhance the model&#x2019;s robustness in complex and changing orchard environments; in-depth exploration of model lightweighting techniques to promote efficient deployment of the model on edge devices, providing stronger and more convenient technical support for the intelligent pest control of the lychee industry (<xref ref-type="bibr" rid="B44">Wang et&#xa0;al., 2025</xref>; <xref ref-type="bibr" rid="B48">Zhang et&#xa0;al., 2025</xref>).</p>
<p>Future work will address dynamic illumination and model compression, while current limits remain modest data, controlled lighting, single-class scope, and regional validation.</p>
<p>Additional Objective Metrics(<xref ref-type="table" rid="T10">
<bold>Table&#xa0;10</bold>
</xref>).</p>
</sec>
</body>
<back>
<sec id="s14" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s15" sec-type="author-contributions">
<title>Author contributions</title>
<p>XW: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. XS:&#xa0;Supervision, Writing &#x2013; review &amp; editing, Funding acquisition, Software, Validation, Data curation, Visualization, Conceptualization, Formal analysis. ZM: Visualization, Supervision, Validation, Writing &#x2013; review &amp; editing, Software. BX: Writing &#x2013; review &amp; editing, Formal analysis, Methodology, Supervision, Data curation, Software, Conceptualization, Resources, Funding acquisition, Validation, Project administration, Visualization, Investigation.</p>
</sec>
<sec id="s16" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec id="s17" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s18" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. The manuscript was grammatically revised and stylistically polished using Kimi K2-0905.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s19" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors&#xa0;and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ali</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Alnajjar</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Jassmi</surname> <given-names>H. A.</given-names>
</name>
<name>
<surname>Gocho</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Serhani</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Performance evaluation of deep CNN-based crack detection and localization techniques for concrete structures</article-title>. <source>Sensors</source> <volume>21</volume>, <elocation-id>1688</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s21051688</pub-id>, PMID: <pub-id pub-id-type="pmid">33804490</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bachhal</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kukreja</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Ahuja</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lilhore</surname> <given-names>U. K.</given-names>
</name>
<name>
<surname>Simaiya</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bijalwan</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Maize leaf disease recognition using PRF-SVM integration: a breakthrough technique</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>10219</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-024-60506-8</pub-id>, PMID: <pub-id pub-id-type="pmid">38702373</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bhatti</surname> <given-names>U. A.</given-names>
</name>
<name>
<surname>Mengxing</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Bazai</surname> <given-names>S. U.</given-names>
</name>
<name>
<surname>Aamir</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Deep Learning for Multimedia Processing Applications: Volume Two: Signal Processing and Pattern Recognition</source>. <edition>1st Edn</edition> (<publisher-loc>Boca Raton</publisher-loc>: <publisher-name>CRC Press</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1201/9781032646268</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>YOLOv4: Optimal speed and accuracy of object detection</article-title>. <source>arXiv:2004.10934</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2004.10934</pub-id>
</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Nanehkaran</surname> <given-names>Y. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Using deep transfer learning for image-based plant disease identification</article-title>. <source>Comput. Electron. Agric.</source> <volume>173</volume>, <elocation-id>105393</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105393</pub-id>
</citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>MoSViT: a lightweight vision transformer framework for efficient disease detection via precision attention mechanism</article-title>. <source>Front. Artif. Intell.</source> <volume>8</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/frai.2025.1498025</pub-id>, PMID: <pub-id pub-id-type="pmid">40206703</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>FD-YOLO11: A feature-enhanced deep learning model for steel surface defect detection</article-title>. <source>IEEE Access</source> <volume>13</volume>, <fpage>63981</fpage>&#x2013;<lpage>63993</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2025.3559733</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ekanayake</surname> <given-names>I. U.</given-names>
</name>
<name>
<surname>Meddage</surname> <given-names>D. P. P.</given-names>
</name>
<name>
<surname>Rathnayake</surname> <given-names>U.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A novel approach to explain the black-box nature of machine learning in compressive strength predictions of concrete using Shapley additive explanations (SHAP)</article-title>. <source>Case Stud. Construct. Mater.</source> <volume>16</volume>, <elocation-id>e01059</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cscm.2022.e01059</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Everingham</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Van Gool</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Williams</surname> <given-names>C. K. I.</given-names>
</name>
<name>
<surname>Winn</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zisserman</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>The pascal visual object classes (VOC) challenge</article-title>. <source>Int. J. Comput. Vision</source> <volume>88</volume>, <fpage>303</fpage>&#x2013;<lpage>338</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11263-009-0275-4</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>C. F.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>L. S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Cattle behavior recognition and tracking method based on improved YOLOv8</article-title>. <source>Trans. Chin. Soc. Agric. Machinery</source> <volume>55</volume>, <fpage>290</fpage>&#x2013;<lpage>301</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2024.05.028</pub-id>
</citation></ref>
<ref id="B11">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Gallagher</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>How to train an ultralytics YOLOv8 oriented bounding box (OBB) model</article-title>. Available online at: <uri xlink:href="https://blog.roboflow.com/train-yolov8-obb-model/">https://blog.roboflow.com/train-yolov8-obb-model/</uri> (Accessed <access-date>October 10, 2025</access-date>).</citation></ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ghayoumi</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Generative adversarial networks in practice</source> (<publisher-loc>Boca Raton London New York</publisher-loc>: <publisher-name>CRC Press, Taylor &amp; Francis Group</publisher-name>).</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Grape leaf disease detection based on attention mechanisms</article-title>. <source>Int. J. Agric. Biol. Eng.</source> <volume>15</volume>, <fpage>205</fpage>&#x2013;<lpage>212</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.25165/j.ijabe.20221505.7548</pub-id>
</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>CTDUNet: A multimodal CNN&#x2013;transformer dual U-shaped network with coordinate space attention for Camellia oleifera pests and diseases segmentation in complex environments</article-title>. <source>Plants</source> <volume>13</volume>, <elocation-id>2274</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/plants13162274</pub-id>, PMID: <pub-id pub-id-type="pmid">39204710</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Mao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Dally</surname> <given-names>W. J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding</article-title>. <source>Fiber</source>. <volume>56</volume> (<issue>4</issue>), <page-range>3&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1510.00149</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Shu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>A method for plant disease enhance detection based on improved YOLOv8</article-title>,&#x201d; in <source>2024 IEEE 33rd International Symposium on Industrial Electronics (ISIE)</source> (<publisher-name>IEEE</publisher-name>, <publisher-loc>Ulsan, Korea, Republic of</publisher-loc>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ISIE54533.2024.10595696</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <source>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-name>IEEE</publisher-name>, <publisher-loc>Las Vegas, NV, USA</publisher-loc>), <fpage>770</fpage>&#x2013;<lpage>778</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id>
</citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hinton</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Vinyals</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Dean</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Distilling the knowledge in a neural network</article-title>. <source>arXiv:1503.02531</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1503.02531</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hoang</surname> <given-names>V.-T.</given-names>
</name>
<name>
<surname>Jo</surname> <given-names>K.-H.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Practical analysis on architecture of EfficientNet</article-title>,&#x201d; in <source>2021 14th International Conference on Human System Interaction (HSI)</source> (<publisher-name>IEEE</publisher-name>, <publisher-loc>Gda&#x144;sk, Poland</publisher-loc>), <fpage>1</fpage>&#x2013;<lpage>4</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/HSI52170.2021.9538782</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hosang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Benenson</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Schiele</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>What makes for effective detection proposals</article-title>? <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>38</volume>, <fpage>814</fpage>&#x2013;<lpage>830</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2015.2465908</pub-id>, PMID: <pub-id pub-id-type="pmid">26959679</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Squeeze-and-excitation networks</article-title>,&#x201d; in <conf-name>2018 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <page-range>7132&#x2013;7141</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2018.00745</pub-id>
</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Isinkaye</surname> <given-names>F. O.</given-names>
</name>
<name>
<surname>Olusanya</surname> <given-names>M. O.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>P. K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep learning and content-based filtering techniques for improving plant disease identification and treatment recommendations: A comprehensive review</article-title>. <source>Heliyon</source> <volume>10</volume>, <elocation-id>e29583</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.heliyon.2024.e29583</pub-id>, PMID: <pub-id pub-id-type="pmid">38737274</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Gouda</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Real-time tracker of chicken for poultry based on attention mechanism-enhanced YOLO-Chicken algorithm</article-title>. <source>Comput. Electron. Agric.</source> <volume>237</volume>, <elocation-id>110640</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2025.110640</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>a). <article-title>Application of YOLOv5 based on attention mechanism and receptive field in identifying defects of thangka images</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>81597</fpage>&#x2013;<lpage>81611</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2022.3195176</pub-id>
</citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>b). <article-title>One-stage disease detection method for maize leaf based on multi-scale feature fusion</article-title>. <source>Appl. Sci.</source> <volume>12</volume>, <elocation-id>7960</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app12167960</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>T. Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Hariharan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Belongie</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Feature pyramid networks for object detection</article-title>,&#x201d; in <conf-name>2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Honolulu, HI, USA</conf-loc>, pp. <page-range>936&#x2013;944</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2017.106</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Path aggregation network for instance segmentation</article-title>,&#x201d; in <source>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-name>IEEE</publisher-name>, <publisher-loc>Salt Lake City, UT</publisher-loc>), <fpage>8759</fpage>&#x2013;<lpage>8768</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2018.00913</pub-id>
</citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>H. X.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>K. B.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y. F.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>S. H.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>W. G.</given-names>
</name>
<name>
<surname>Gou</surname> <given-names>J. P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Research on lightweight plant identification model based on improved YOLOv5s</article-title>. <source>Trans. Chin. Soc. Agric. Machinery</source> <volume>54</volume>, <fpage>267</fpage>&#x2013;<lpage>276</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2023.08.026</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Nelson.</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Solawetz.</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>).<article-title>YOLOv5 is here: state-of-the-art object detection at 140 FPS</article-title>. Available online at: <uri xlink:href="https://blog.roboflow.com/yolov5-is-here/">https://blog.roboflow.com/yolov5-is-here/</uri> (Accessed <access-date>October 10, 2025</access-date>).</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Lightweight marine biodetection model based on improved YOLOv10</article-title>. <source>Alexandria Eng. J.</source> <volume>119</volume>, <fpage>379</fpage>&#x2013;<lpage>390</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aej.2025.01.077</pub-id>
</citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>How do vision transformers work</article-title>? <source>arXiv:2202.06709</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2202.06709</pub-id>
</citation></ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Qian</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ning</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>MobileNetV3 for image classification</article-title>,&#x201d; in <source>2021 IEEE 2nd International Conference on Big Data, Artificial Intelligence and Internet of Things Engineering (ICBAIE)</source> (<publisher-name>IEEE</publisher-name>, <publisher-loc>Nanchang, China</publisher-loc>), <fpage>490</fpage>&#x2013;<lpage>497</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICBAIE52039.2021.9389905</pub-id>
</citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramamurthy</surname> <given-names>K.</given-names>
</name>
<name>
<surname>M.</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Anand</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Mathikshara</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Johnson</surname> <given-names>A.</given-names>
</name>
<name>
<surname>XXXR.</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Attention embedded residual CNN for disease detection in tomato leaves</article-title>. <source>Appl. Soft Comput.</source> <volume>86</volume>, <elocation-id>105933</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.asoc.2019.105933</pub-id>
</citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>YOLOv3: An incremental improvement</article-title>. <source>arXiv:1804.02767</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv:1804.02767</pub-id>
</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ridnik</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ben-Baruch</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Noy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zelnik-Manor</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>ImageNet-21K pretraining for the masses</article-title>. <source>arXiv.2104.10972</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv:2104.10972</pub-id>
</citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ryo</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Explainable artificial intelligence and interpretable machine learning for agricultural data analysis</article-title>. <source>Artif. Intell. Agric</source>. <volume>6</volume>, <page-range>257&#x2013;265</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2139/ssrn.4230728</pub-id>
</citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sahu</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Chug</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>A. P.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>a). <article-title>Classification of crop leaf diseases using image to image translation with deep-dream</article-title>. <source>Multimed. Tools Appl.</source> <volume>82</volume>, <fpage>35585</fpage>&#x2013;<lpage>35619</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-023-14994-x</pub-id>
</citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sangjan</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Carter</surname> <given-names>A. H.</given-names>
</name>
<name>
<surname>Pumphrey</surname> <given-names>M. O.</given-names>
</name>
<name>
<surname>Jitkov</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Sankaran</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Development of a raspberry pi-based sensor system for automated in-field monitoring to support crop breeding programs</article-title>. <source>Inventions</source> <volume>6</volume>, <elocation-id>42</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/inventions6020042</pub-id>
</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Mei</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>LDDFSF-YOLO11: A lightweight insulator defect detection method focusing on small-sized features</article-title>. <source>IEEE Access</source> <volume>13</volume>, <fpage>90273</fpage>&#x2013;<lpage>90292</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2025.3569970</pub-id>
</citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoaib</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>B.</given-names>
</name>
<name>
<surname>EI-Sappagh</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ullah</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Alenezi</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>An advanced deep learning models-based plant disease detection: A review of recent research</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2023.1158933</pub-id>, PMID: <pub-id pub-id-type="pmid">37025141</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srinivas</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>T. Y.</given-names>
</name>
<name>
<surname>Parmar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Shlens</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Abbeel</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Vaswani</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Bottleneck transformers for visual recognition</article-title>. <source>arXiv:2101.11605</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2101.11605</pub-id>
</citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srinivasu</surname> <given-names>P. N.</given-names>
</name>
<name>
<surname>Kumari</surname> <given-names>G. L. A.</given-names>
</name>
<name>
<surname>Narahari</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Alhumam</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Exploring the impact of hyperparameter and data augmentation in YOLO V10 for accurate bone fracture detection from X-ray images</article-title>. <source>Sci. Rep.</source> <volume>15</volume>, <fpage>9828</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-025-93505-4</pub-id>, PMID: <pub-id pub-id-type="pmid">40119100</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tahir</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Qiblawey</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Khandakar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Khurshid</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Musharavati</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Deep learning for reliable classification of COVID-19, MERS, and SARS from chest X-ray images</article-title>. <source>Cognit. Comput.</source> <volume>14</volume>, <fpage>1752</fpage>&#x2013;<lpage>1772</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12559-021-09955-1</pub-id>, PMID: <pub-id pub-id-type="pmid">35035591</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>Z</given-names>
</name>
</person-group>. (<year>2025</year>). <article-title>DSS-YOLO: an improved lightweight real-time fire detection model based on YOLOv8</article-title>. <source>Sci. Rep.</source> <volume>15</volume>, <fpage>8963</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-025-93278-w</pub-id>, PMID: <pub-id pub-id-type="pmid">40089566</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>J. P.</given-names>
</name>
<name>
<surname>He</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhen</surname> <given-names>Q. G.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H. P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Static and dynamic detection and counting method for camellia oleifera fruits based on improved COF-YOLOv8n</article-title>. <source>Trans. Chin. Soc. Agric. Machinery</source> <volume>55</volume>, <fpage>193</fpage>&#x2013;<lpage>203</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2024.04.019</pub-id>
</citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>C2PSA-enhanced YOLOv11 architecture: A novel approach for small target detection in cotton disease diagnosis</article-title>. <source>arXiv:2406.19407</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2508.12219</pub-id>
</citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xue</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Experiment study on UAV target detection algorithm based on YOLOv8n-ACW</article-title>. <source>Sci. Rep.</source> <volume>15</volume>, <fpage>11352</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-025-91394-1</pub-id>, PMID: <pub-id pub-id-type="pmid">40175443</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>YOLOv8 forestry pest recognition based on improved re-parametric convolution.Front</article-title>. <source>Plant Sci.</source> <volume>16</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2025.1552853</pub-id>, PMID: <pub-id pub-id-type="pmid">40134619</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A series of methods incorporating deep learning and computer vision techniques in the study of fruit fly (Diptera: Tephritidae) regurgitation</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2023.1337467</pub-id>, PMID: <pub-id pub-id-type="pmid">38288408</pub-id></citation></ref>
</ref-list>
</back>
</article>