<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2024.1348402</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Identification of cotton pest and disease based on CFNet- VoV-GCSP -LSKNet-YOLOv8s: a new era of precision agriculture</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Li</surname><given-names>Rujia</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2594076"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>He</surname><given-names>Yiting</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname><given-names>Yadong</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2639288"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Qin</surname><given-names>Weibo</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2617804"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Abbas</surname><given-names>Arzlan</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1241363"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ji</surname><given-names>Rongbiao</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname><given-names>Shuang</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname><given-names>Yehui</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sun</surname><given-names>Xiaohai</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yang</surname><given-names>Jianping</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>*</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Big Data, Yunnan Agricultural University</institution>, <addr-line>Kunming</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>College of Plant Protection, Jilin Agricultural University</institution>, <addr-line>Changchun</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Jilin Haicheng Technology Co., Ltd.</institution>, <addr-line>Changchun</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Yuriy L. Orlov, I.M.Sechenov First Moscow State Medical University, Russia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Yunchao Tang, Guangxi University, China</p>
<p>Aibin Chen, Central South University Forestry and Technology, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Jianping Yang, <email xlink:href="mailto:yangjp@ynau.edu.cn">yangjp@ynau.edu.cn</email>
</p>
</fn>
<fn fn-type="equal" id="fn002">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>20</day>
<month>02</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1348402</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>01</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Li, He, Li, Qin, Abbas, Ji, Li, Wu, Sun and Yang</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Li, He, Li, Qin, Abbas, Ji, Li, Wu, Sun and Yang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The study addresses challenges in detecting cotton leaf pests and diseases under natural conditions. Traditional methods face difficulties in this context, highlighting the need for improved identification techniques.</p>
</sec>
<sec>
<title>Methods</title>
<p>The proposed method involves a new model named CFNet-VoV-GCSP-LSKNet-YOLOv8s. This model is an enhancement of YOLOv8s and includes several key modifications: (1) CFNet Module. Replaces all C2F modules in the backbone network to improve multi-scale object feature fusion. (2) VoV-GCSP Module. Replaces C2F modules in the YOLOv8s head, balancing model accuracy with reduced computational load. (3) LSKNet Attention Mechanism. Integrated into the small object layers of both the backbone and head to enhance detection of small objects. (4) XIoU Loss Function. Introduced to improve the model's convergence performance.</p>
</sec>
<sec>
<title>Results</title>
<p>The proposed method achieves high performance metrics: Precision (P), 89.9%. Recall Rate (R), 90.7%. Mean Average Precision (mAP@0.5), 93.7%. The model has a memory footprint of 23.3MB and a detection time of 8.01ms. When compared with other models like YOLO v5s, YOLOX, YOLO v7, Faster R-CNN, YOLOv8n, YOLOv7-tiny, CenterNet, EfficientDet, and YOLOv8s, it shows an average accuracy improvement ranging from 1.2% to 21.8%.</p>
</sec>
<sec>
<title>Discussion</title>
<p>The study demonstrates that the CFNet-VoV-GCSP-LSKNet-YOLOv8s model can effectively identify cotton pests and diseases in complex environments. This method provides a valuable technical resource for the identification and control of cotton pests and diseases, indicating significant improvements over existing methods.</p>
</sec>
</abstract>
<kwd-group>
<kwd>artificial intelligence</kwd>
<kwd>cotton</kwd>
<kwd>pests and diseases</kwd>
<kwd>deep learning</kwd>
<kwd>machine learning</kwd>
<kwd>XIoU</kwd>
<kwd>YOLO</kwd>
</kwd-group>
<counts>
<fig-count count="12"/>
<table-count count="5"/>
<equation-count count="12"/>
<ref-count count="30"/>
<page-count count="14"/>
<word-count count="6278"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Technical Advances in Plant Science</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Cotton is one of the vital fiber crops in China, extensively employed in textile production and the manufacturing of cotton goods. However, the cotton industry in China has been severely jeopardized by the pervasive threats of diseases and pests, leading to adverse impacts on the yield (<xref ref-type="bibr" rid="B4">Chohan et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B1">Abbas et&#xa0;al., 2021</xref>). Traditional methods of pest and disease detection often rely on seasoned experts who gauge the health of cotton leaves through visual inspection. Despite its widespread use, this conventional approach suffers from multiple shortcomings. First, these methods are labor-intensive and time-consuming, requiring significant human resources, especially in large-scale cotton cultivation. Second, the manual inspections depend on the subjective assessments of experts, introducing variability and compromising the consistency and accuracy of the results.</p>
<p>With the advent of advancements in computer vision technology and deep learning algorithms (<xref ref-type="bibr" rid="B22">Wang C. et&#xa0;al., 2023</xref>), the agricultural sector has witnessed new avenues for pest and disease detection (<xref ref-type="bibr" rid="B14">Meng et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B26">Ye et&#xa0;al., 2023</xref>). These technologies not only automate the identification process but also enhance the speed and accuracy of detections. Notably, the YOLO (You Only Look Once) algorithm (<xref ref-type="bibr" rid="B7">Jiang et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B30">Zhang Y. et&#xa0;al., 2023</xref>) has achieved remarkable success in this context, acclaimed for its real-time processing, multi-scale support, automation, and efficient data handling, thus providing a robust tool for pest and disease monitoring and management in agriculture.</p>
<p>Several researchers have made notable advancements in the field of cotton disease identification and monitoring. Caldeira, R. F. et&#xa0;al (<xref ref-type="bibr" rid="B2">Caldeira et&#xa0;al., 2021</xref>)used the convolutional neural network learning models GoogleNet and Resnet50 to monitor the health status of cotton crops, and obtained accuracy rates of 86.6% and 89.2% respectively.</p>
<p>Nannan Zhang et&#xa0;al. (<xref ref-type="bibr" rid="B15">Nannan et&#xa0;al., 2020</xref>) presented the CBAM-YOLO v7 algorithm, an improved attention mechanism YOLO v7, with a mAP of 85.5%, providing a strong theoretical foundation for real-time cotton leaf disease monitoring. Yuanjia <xref ref-type="bibr" rid="B28">Zhang et&#xa0;al. (2022a)</xref> developed a real-time, high-performance detection model based on an enhanced YOLOX algorithm. The comparative results also demonstrated that the improved model achieved mAP values 11.50%, 21.17%, 9.34%, 10.22%, and 8.33% higher than the other five algorithms, meeting real-time speed detection requirements. According to <xref ref-type="bibr" rid="B12">Liu and Wang (2020)</xref>, the feature layer of the Yolo V3 model using an image pyramid to achieve multi-scale feature detection, resulting in improved accuracy and speed for the detection of diseases and pests in tomatoes. Zhenyang <xref ref-type="bibr" rid="B25">Xue et&#xa0;al. (2023)</xref> proposed YOLO-Tea, an enhanced model based on You Only Look Once version 5 (YOLOv5), outperforming YOLOv5s by 0.3% to 15.0% across all&#xa0;test data. Furthermore, <xref ref-type="bibr" rid="B13">Liu et&#xa0;al. (2023)</xref> introduced MRF-YOLO, a deep learning method with multi-receptive field extraction based on YOLOX, integrating a small target detection layer to enhance precision. <xref ref-type="bibr" rid="B6">Jajja et&#xa0;al. (2022)</xref> proposed a Compact Convolutional Transformer (CCT)-based approach is to classify the image dataset, achieving an impressive accuracy of 97.2% and proving its effectiveness compared to state-of-the-art approaches. Additionally, <xref ref-type="bibr" rid="B16">Patil and Patil (2021)</xref> developed a deep CNN model that accurately collected images throughout the complete process of training and validation in image pre-processing, ensuring high efficiency and accuracy for cotton disease detection. Liang, X (<xref ref-type="bibr" rid="B11">Liang, 2021</xref>) proposed a metric learning method for extraction and classification of cotton leaf spot characteristics. By constructing a metric space and using KNN as a point classifier, common models such as Vgg, DenseNet and ResNet were compared. The spatial structure optimizer (SSO) is introduced to perform local optimization of the model. Experimental results show that the average classification accuracy of S-DenseNet is 7.7% higher than the other two networks, and DenseNet shows the highest classification accuracy. Tao, Y et&#xa0;al (<xref ref-type="bibr" rid="B19">Tao et&#xa0;al., 2022</xref>). proposed an automatic detection method for cotton diseases, using ConvNeXt to combine the convolutional neural network architecture with the inherent advantages of Transformer. The Multi-Scale Spatial Pyramid Attention (MSPA) module can help ConvNeXt focus on important areas of feature maps. The results show that the model performs well in terms of recognition accuracy and detection speed.</p>
<p>In the realm of identifying pests, diseases, and behaviors using YOLO algorithms, extensive research has been conducted, highlighting their current significance. However, when applied to cotton pests and diseases identification, conventional YOLO algorithms encounter challenges in detecting cotton leaf diseases under natural conditions, difficulty in extracting features from small targets, and low efficiency (<xref ref-type="bibr" rid="B20">Terven and Cordova-Esparza, 2023</xref>). To overcome these challenges, this study presents an enhanced method for cotton peat and disease identification, built upon YOLOv8s (<xref ref-type="bibr" rid="B24">Xie and Sun, 2023</xref>). This method involves replacing the C2F modules in the backbone network with CFNet modules (<xref ref-type="bibr" rid="B27">Zhang G. et&#xa0;al., 2023</xref>) and substituting all C2F modules in the YOLOv8s header with VoV-GCSP modules (<xref ref-type="bibr" rid="B8">Li et&#xa0;al., 2022</xref>). It also integrates the LSKNet attention mechanism (<xref ref-type="bibr" rid="B10">Li et&#xa0;al., 2023</xref>) into the small target layers of both the backbone network and header. Furthermore, the XIoU loss function is introduced to streamline the model while preserving accuracy, ultimately enhancing the model&#x2019;s convergence performance.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Experimental data</title>
<p>The data used in this study were sourced from six publicly available cotton pest and disease datasets on KAGGLE (<ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/saeedazfar/customized-cotton-disease-dataset">https://www.kaggle.com/datasets/saeedazfar/customized-cotton-disease-dataset</ext-link>; <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/paridhijain02122001/cotton-crop-disease-detection">https://www.kaggle.com/datasets/paridhijain02122001/cotton-crop-disease-detection</ext-link>). Images that were blurry or had indistinct features were removed during data cleaning, resulting in a total of 4,703 images for pests and diseases, as illustrated in <xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>. Due to data imbalance, data augmentation techniques such as rotation, brightness adjustment, and random cropping were applied (<xref ref-type="bibr" rid="B18">Tang et&#xa0;al., 2020</xref>), expanding the dataset to 5,927 images. The training and test datasets were then divided in an 8:2 ratio using random sampling, as shown in <xref ref-type="table" rid="T1"><bold>Table&#xa0;1</bold></xref> below. During training, the image size was set to 640&#xd7;640 pixels.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Examples of Images of <bold>(A)</bold> Nocturnal moth larvae <bold>(B)</bold> Cotton angular leaf spot <bold>(C)</bold> cotton boll rot <bold>(D)</bold> Cotton hoarfrost <bold>(E)</bold> Health and <bold>(F)</bold> Alternaria leaf spot of cotton.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g001.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Information on cotton pest and disease data sets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Pest and disease categories</th>
<th valign="middle" align="center">Original data quantity</th>
<th valign="middle" align="center">Quantity after expansion</th>
<th valign="middle" align="center">Label</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Army worm</td>
<td valign="middle" align="center">799</td>
<td valign="middle" align="center">799</td>
<td valign="middle" align="center">Army_worm</td>
</tr>
<tr>
<td valign="middle" align="center">Bacterial Blight</td>
<td valign="middle" align="center">1136</td>
<td valign="middle" align="center">1136</td>
<td valign="middle" align="center">Bacterial_Blight_edited</td>
</tr>
<tr>
<td valign="middle" align="center">Cotton Boll Rot</td>
<td valign="middle" align="center">916</td>
<td valign="middle" align="center">916</td>
<td valign="middle" align="center">Cotton_Boll_Rot</td>
</tr>
<tr>
<td valign="middle" align="center">Diseased cotton leaf</td>
<td valign="middle" align="center">340</td>
<td valign="middle" align="center">1020</td>
<td valign="middle" align="center">diseased_cotton_leaf</td>
</tr>
<tr>
<td valign="middle" align="center">Healthy</td>
<td valign="middle" align="center">968</td>
<td valign="middle" align="center">968</td>
<td valign="middle" align="center">Healthy</td>
</tr>
<tr>
<td valign="middle" align="center">Target spot</td>
<td valign="middle" align="center">544</td>
<td valign="middle" align="center">1088</td>
<td valign="middle" align="center">Target_spot</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Cotton pest and disease identification</title>
<sec id="s2_2_1">
<label>2.2.1</label>
<title>YOLOv8 network structure</title>
<p>YOLOv8 is a state-of-the-art (SOTA) model that builds upon the successes of previous YOLO versions, incorporating novel features and enhancements to further improve performance and versatility. Specific innovations include a new backbone network, a new Anchor-Free detection head, and a novel loss function. Anchor-Free Detection Head: Traditional object detection models utilize anchor boxes to determine the position and size of targets. In contrast, the Anchor-Free detection head learns the key-points or bounding boxes of the targets, thus eliminating the need for anchor boxes. This approach enables the model to better adapt to targets of varying sizes and shapes while reducing the complexity associated with tuning anchor boxes. Novel Loss Function: The loss function serves as feedback during training, assisting the model in fine-tuning its parameters for better target approximation. Currently, the YOLOv8 series has introduced five different versions, namely YOLOv8n, YOLOv8s, YOLOv8m, YOLOv8l, and YOLOv8x. The model&#x2019;s parameter and computational complexity increase with the depth and width of the model. Users can choose the appropriate network structure based on their application scenarios. The YOLOv8s version employs a lighter network structure and fewer training data, aiming to maintain relatively fast detection speed and high accuracy while efficiently deploying on embedded devices and small applications (<xref ref-type="bibr" rid="B23">Wang G. et&#xa0;al., 2023</xref>). This makes YOLOv8s an ideal choice for real-time object detection applications. Therefore, this paper adopts the YOLOv8s model to meet the demand for efficient object detection. The YOLOv8 model detection network structure, as illustrated in <xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref> below, comprises the Backbone, FPN, and Head.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>YOLOv8s Network Architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g002.tif"/>
</fig>
<p>The Backbone serves as YOLOv8&#x2019;s primary feature extraction network. Images fed into this network initially undergo feature extraction to produce what is commonly referred to as feature layers, a comprehensive set of features derived from the input images. These Feature Pyramid Network (FPN) in YOLOv8 is an augmented feature extraction component. Three significant feature layers obtained from the backbone network are further integrated in this section. The objective of this feature fusion is to combine feature information from various scales. The FPN continues to extract features from the already obtained significant feature layers. YOLOv8 still employs the Panet architecture, which not only up-samples the features for fusion but also down-samples them for an additional fusion. The Head in YOLOv8 serves as the classifier and regressor. Through the Backbone and FPN, we can obtain three enhanced, significant feature layers.</p>
</sec>
<sec id="s2_2_2">
<label>2.2.2</label>
<title>CFNet-VoV-GCSP-LSKNet-YOLOv8s network structure</title>
<p>The network structure of the cotton pest and disease identification model based on CFNet-VoV-GCSP-LSKNet-YOLOv8s proposed in this paper is illustrated in <xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3</bold></xref>.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>CFNet -VoV-GCSP-LSKNet-YOLOv8s network structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g003.tif"/>
</fig>
<p>YOLOv8 is a state-of-the-art (SOTA) model, further categorized into YOLOv8n, YOLOv8s, YOLOv8m, YOLOv8l, and YOLOv8x. The YOLOv8 detection network structure consists of Backbone, FPN, and Head. While the Backbone has a large number of parameters and a long training time, it falls short in the detection of small objects. To address these limitations, we propose a cotton pest and disease identification model based on CFNet-VoV-GCSP-LSKNet-YOLOv8s. Firstly, the CFNet module replaces all C2F modules in the YOLOv8s Backbone. Then, a feature integration operation is inserted in the Backbone, effectively utilizing a large proportion of the Backbone to fuse multi-scale features, thereby improving the model&#x2019;s recognition rate. Secondly, in the Head of YOLOv8s, the VoV-GCSP module replaces all C2F modules, enhancing the features extracted by the Backbone while also reducing the model size without sacrificing accuracy. Additionally, the LSKNet attention mechanism is incorporated into both the Backbone and Head to improve the detection of small objects. Lastly, the XIoU loss function is introduced to enhance model convergence, thereby achieving accurate identification of cotton pests and diseases.</p>
</sec>
<sec id="s2_2_3">
<label>2.2.3</label>
<title>Cascaded fusion network</title>
<p>In the YOLOv8 Backbone, the C2F module attempts to fuse shallow feature maps with high resolution but limited semantic information with deep feature maps that have low resolution but rich semantic content. However, we argue that this approach might be insufficient for effective multi-scale feature fusion, especially when compared to heavy classification backbones where the parameters allocated for feature fusion are limited. To address this issue, we propose a new architecture named Cascaded Fusion Network (CFNet). Apart from the initial high-resolution feature-extracting Backbone and several blocks, we introduce multiple cascading stages to generate multi-scale features within CFNet. Each stage consists of a sub-backbone for feature extraction and an extremely lightweight transformation block for feature integration. This design allows for a more in-depth and effective fusion of features, leveraging a large proportion of the Backbone&#x2019;s parameters. By replacing all C2F modules in the Backbone with CFNet and then inserting feature integration operations, we achieve effective fusion of multi-scale features across a significant portion of the Backbone.</p>
<p>The core design philosophy of CFNet involves introducing multiple cascading stages, each stage consisting of a specialized feature extraction sub-backbone and an extremely lightweight transformation block, effectively capturing and merging multi-scale features from fine-grained to coarse-grained. These cascading stages not only process the output from the previous stage but also deeply interact with the corresponding features of the main backbone network, achieving complex feature integration. The feature maps produced by each cascading stage are optimized and merged through specifically designed transformation blocks, enhancing the model&#x2019;s ability to represent features. This structure is especially suitable for object detection tasks that require efficient multi-scale feature fusion. By optimizing and deeply integrating features at different levels, CFNet improves model performance while maintaining relatively low computational costs.</p>
<p>Suppose <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the output feature map of the ith cascading stage of the input image, where i represents the sequence number of the cascading stage. F represents the feature extraction function, and T represents the feature transformation function (lightweight transformation block). Thus, each cascading stage can be formally represented as shown in <xref ref-type="disp-formula" rid="eq1"><bold>Equation 1</bold></xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mn>i+1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;for&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;i</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In CFNet, M represents the total number of cascading stages. <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the initial high-resolution feature map (as shown in <xref ref-type="disp-formula" rid="eq2"><bold>Equation 2</bold></xref>). For feature fusion, suppose <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the j fused feature map, corresponding to different spatial resolutions, such as <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>P</mml:mi>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>P</mml:mi>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> etc. Each <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be calculated through the output of the cascade and the corresponding transformation function <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>M</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mn>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;for&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;j=3,4,5</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In CFNet,. refers to the output of the final cascading stage. The network architecture of CFNet is illustrated in <xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4</bold></xref>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>CFNet network architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g004.tif"/>
</fig>
<p>The CFNet architecture commences by inputting an image with spatial dimensions of H &#xd7; W through a neck and N successive blocks, extracting high-resolution features with dimensions of <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mfrac>
<mml:mi>H</mml:mi>
<mml:mn>4</mml:mn>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mi>W</mml:mi>
<mml:mn>4</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>. These features are subsequently directed into M cascaded stages for the extraction of multi-scale features. Prior to entry into the M cascaded stages, the extracted high-resolution features are downscaled using a 2&#xd7;2 convolution kernel with a stride of 2. The network&#x2019;s architecture at each stage maintains a consistent structural format but varies in scale, comprising differing numbers of processing blocks. Each stage consists of a sub-backbone network and an ultra-lightweight transition block, both dedicated to the extraction and integration of features. For clarity, the assemblage of blocks within each stage, addressing features of the same scale, is termed a block group. The three block groups within the ith stage encompass <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msubsup>
<mml:mtext>n</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msubsup>
<mml:mtext>n</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msubsup>
<mml:mtext>n</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mn>3</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> blocks, respectively. In the final block group of each stage, a so-called focal block is implemented to enhance feature processing. Each stage outputs features P3, P4, P5 with strides of 8, 16, 32, respectively, of which only P3 features are utilized for input into the subsequent stage. In the network&#x2019;s final stage, features P3, P4, and P5 are amalgamated, serving dense prediction tasks. By substituting all C2F modules in the backbone with CFNet and incorporating feature integration operations, effective fusion of multi-scale features is achieved throughout a significant portion of the backbone. <xref ref-type="fig" rid="f5"><bold>Figures&#xa0;5</bold></xref> and <xref ref-type="fig" rid="f6"><bold>6</bold></xref> provide more details about transition blocks and focus blocks.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Transition block.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g005.tif"/>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Focal NeXt block.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g006.tif"/>
</fig>
<p>As depicted in <xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>, When given C<sub>3</sub>, C<sub>4</sub>, and C<sub>5</sub> as inputs, the transition block produces outputs P<sub>3</sub> and P<sub>4</sub>. The term &#x201c;Conv -dx&#x201d; refers to a 1&#xd7;1 convolution operation that outputs a channel number of dx, where dx matches the channel number of the input feature C<sub>x</sub>. The circles marked with &#x201c;+&#x201d; and &#x201c;C&#x201d; represent element-wise addition and concatenation operations, respectively. Additionally, the notation &#x201c;2x&#x201d; is used to indicate the upsampling of features by a scaling factor of 2.</p>
<p>This design facilitates the effective integration of multi-scale features. By adjusting the channel dimensions through 1&#xd7;1 convolutions and managing the spatial resolutions via addition, concatenation, and upsampling operations, the transition block efficiently processes the varied scales of the input features (C<sub>3</sub>, C<sub>4</sub>, C<sub>5</sub>) and transforms them into the desired output formats (P<sub>3</sub>, P<sub>4</sub>), which are then suitable for subsequent stages of the network&#x2019;s processing pipeline.</p>
<p>As shown in <xref ref-type="fig" rid="f6"><bold>Figure&#xa0;6</bold></xref>, N is the number of channels of the output feature. d7&#xd7;7 represents the 7&#xd7;7 depth convolution, a7&#xd7;7 represents the window size, and R is the expansion rate of the additional convolution. GELU is the activation function. Each d7&#xd7;7 or a7&#xd7;7 is followed by a LayerNorm layer and a GELU unit. This paper proposes a novel focus block to enlarge the receptive fields of neurons in the last block group of each stage as an effective alternative strategy. The design of the focus module introduces extended depth convolution and two skip connections in the ConvNeXt module, thereby achieving the integration of fine-grained local interaction and coarse-grained global interaction.</p>
</sec>
<sec id="s2_2_4">
<label>2.2.4</label>
<title>VoV-GCSP network structure</title>
<p>The YOLOv8 network employs a substantial number of C2F modules in its neck for feature extraction. However, this structure results in an increase in computational complexity and the number of parameters, leading to significant time consumption. Lightweight networks like Exception and ShuffleNet address the time-consuming issue of standard convolutions by utilizing depth-wise separable convolutions (DSC), albeit at the cost of sacrificing accuracy. The GSConv convolution module is an innovative approach that combines Standard Convolution (SC), Depth-Wise Convolution (DWConv), and channel shuffle operations. The core idea involves partitioning the input channels into multiple groups, performing independent depth-wise separable convolutions on each group to reduce computational complexity. This design aims to mitigate the issue of low recognition accuracy due to insufficient feature extraction and fusion capabilities. The groups are then recombined through channel shuffling. GSConv combines SC, DSC, and Shuffle, exhibiting performance similar to SC but with lower computational costs. The depth layer calculation is shown in <xref ref-type="disp-formula" rid="eq3"><bold>Equation 3</bold></xref>, and the GSConv layer calculation is shown in <xref ref-type="disp-formula" rid="eq4"><bold>Equation 4</bold></xref>.</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mtext>DSC</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>H</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>m</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>n</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>P</mml:mtext>
<mml:mrow>
<mml:mtext>out</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtext>GSConv</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mtext>W</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>H</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>m</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>n</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mtext>P</mml:mtext>
<mml:mrow>
<mml:mtext>out</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>P</mml:mtext>
<mml:mrow>
<mml:mtext>out</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>W and H represent the width and height of the feature map, respectively, and m&#xd7;n is the size of the convolution kernel. Pin and Pout represent input and output function channel numbers. In scenarios where the input feature channel count escalates, the computational demand of the GSConv convolution diminishes, yet it retains a feature extraction proficiency analogous to its contemporaries. The integration of GSConv has been instrumental in the strategic simplification of the model&#x2019;s complexity. To augment the inference velocity of the network model, while concurrently preserving its precision in detection, we have implemented the VoV-GSCSP module, building upon the foundational GSConv module. The VoV-GSCSP represents a sophisticated hybrid network architecture, which skillfully merges the attributes of GSConv with the essence of VoVNet, supplemented by the incorporation of (Squeeze-and-Excitation, SE) blocks. This architectural design is meticulously tailored to enhance both the quality and efficiency of feature extraction. By segmenting the convolutional layers into discrete groups, the Grouped Separable Convolution effectively minimizes the parameter count and computational complexity. The Squeeze-and-Excitation blocks intensify the network&#x2019;s representational prowess by concentrating on salient channel features. This innovative structural design endows the VoV-GSCSP module with the capability to sustain high computational efficiency while simultaneously elevating the feature representation and overall performance of the network. The module, engineered with a one-off aggregation methodology, markedly amplifies the inference speed of the network model, all the while maintaining its superior detection accuracy. The configurations of the GSConv convolution and the VoV-GCSP network are exemplified in <xref ref-type="fig" rid="f7"><bold>Figure&#xa0;7</bold></xref> and <xref ref-type="fig" rid="f8"><bold>Figure&#xa0;8</bold></xref>.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>GSConv.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g007.tif"/>
</fig>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>VoV-GSCSP.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g008.tif"/>
</fig>
</sec>
<sec id="s2_2_5">
<label>2.2.5</label>
<title>Attention mechanism LSK</title>
<p>Attention mechanisms serve as a straight forward effective approach yet to enhance neural representations. Channel attention modules like SE blocks utilize global average information to re-weight feature channels, while spatial attention modules such as GENet, GCNet, and SGE enhance the network&#x2019;s capability to model contextual information via spatial masks. Techniques like CBAM and BAM amalgamate channel and spatial attentions, leveraging the strengths of both. Beyond channel/spatial attention mechanisms, kernel selection is another adaptive and effective technique for dynamic contextual modeling. LSKNet is designed based on attention mechanisms and kernel selection technologies to better model the features of different targets in remote sensing scenarios. It also boasts advantages like relatively fewer parameters and computational complexity, thereby facilitating improved computational efficiency and speed in practical applications. LSKNet is a novel neural network architecture specifically aimed at remote sensing object detection tasks. It enhances contextual modeling and feature extraction through selective mechanisms and adaptive spatial aggregation, consequently improving the performance in small object detection. A detailed structural comparison is shown in <xref ref-type="fig" rid="f9"><bold>Figure&#xa0;9</bold></xref>.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Conceptual diagram of the LSK module.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g009.tif"/>
</fig>
</sec>
<sec id="s2_2_6">
<label>2.2.6</label>
<title>XIoU</title>
<p>The Loss Function is a metric that measures the difference between the predicted values of a model and the actual values. During training, the model attempts to minimize the value of the loss function to improve its accuracy. YOLOv8s adopts the CIoU loss function, composed of position, confidence, and class functions. This traditional loss function generally relies on the aggregation of bounding box regression indicators, without considering the mismatch in direction between the required ground truth boxes and predicted boxes, leading to slow convergence and low efficiency. The XIoU loss function plays a crucial role in object detection tasks by emphasizing varying degrees of overlap between targets. By combining the regression of predicted boxes with real boxes, this loss function prevents issues such as overlapping center points and identical aspect ratios that would degrade into the IOU loss function. This ensures the effective completion of boundary box regression, improving the robustness of the bounding boxes. Therefore, in this study, the XIoU loss function is introduced as an improvement to the model. Compared to the original CIoU loss function, the penalty term gradient of XIoU is smoother, resulting in smaller regression errors and better regression performance. It also effectively enhances the recognition accuracy of cotton leaf diseases and pests.</p>
<p>XIoU calculation formula is as shown in <xref ref-type="disp-formula" rid="eq5"><bold>Equation 5</bold></xref>:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>&#x3c5;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>&#x3bd;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bd;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>&#x3bd;</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The penalty term is defined as shown in <xref ref-type="disp-formula" rid="eq8"><bold>Equation 8</bold></xref>:</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x211c;</mml:mtext>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>&#x3c5;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>As shown in <xref ref-type="disp-formula" rid="eq5"><bold>Equations 5</bold></xref>&#x2013;<xref ref-type="disp-formula" rid="eq8"><bold>8</bold></xref>, IoU stands for the traditional regression loss. <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the squared Euclidean distance between the two rectangular bounding boxes. <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the square of the diagonal distance between two rectangular boxes. b and <italic>b<sup>gt</sup>
</italic> denote the central points of the two bounding boxes. <inline-formula>
<mml:math display="inline" id="im13">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula> weight coefficient.<italic>v</italic> is used to measure the consistency of the relative proportions between the two boxes. <italic>w<sup>gt</sup>
</italic>, <italic>h<sup>gt</sup>
</italic>, <italic>w</italic> and <italic>h</italic> are&#xa0;the width and height of the two boxes, respectively. The primary goal of XIoU is to improve the IoU metric by considering the intersection area between the boxes, offering a better representation of their overlap. The parameter <italic>&#x3b1;</italic> is used to adjust the difference between XIoU and IoU, thereby reflecting the similarity between the boxes more accurately and accelerating the network&#x2019;s convergence.</p>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="results|discussion">
<label>3</label>
<title>Results and discussion</title>
<sec id="s3_1">
<label>3.1</label>
<title>Improve model identification results and analysis</title>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Experimental setup and evaluation metrics</title>
<p>The model was trained using the PyTorch framework on a laboratory server equipped with an Intel Core i9-10900KF processor, 16 GB of CPU memory, and an NVIDIA GeForce RTX 3080 GPU. The operating environment was Windows 10, with Python 3.8, PyTorch 1.11.0, and CUDA 13.0 used for algorithmic optimization. Training parameters included 150&#xa0;epochs, a batch size of 8, and an image input resolution of 640&#xd7;640 pixels. All other settings were kept at their default values.</p>
<p>Performance metrics used for model evaluation included Precision (P), Recall (R), Mean Average Precision (mAP), and model size. Precision is defined as the fraction of true positives among the predicted positives, while Recall measures the fraction of actual positives correctly identified by the model. Mean Average Precision (mAP) serves as a comprehensive performance metric. The above indicators such as <xref ref-type="disp-formula" rid="eq9"><bold>Equations 9</bold></xref>&#x2013;<xref ref-type="disp-formula" rid="eq12"><bold>12</bold></xref> shown.</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mtext>dr</mml:mtext>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mtext>n</mml:mtext>
</mml:munderover>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the equations, TP represents the number of true positives, FP stands for false positives, and FN signifies false negatives.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Cotton pest and disease recognition results</title>
<p>To validate the superior performance of the proposed CFNet- VoV-GCSP -LSKNet -YOLOv8 architecture for the identification of six types of cotton pests and diseases, we compared our model with the original YOLOv8s algorithm, as shown in <xref ref-type="table" rid="T2"><bold>Table&#xa0;2</bold></xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison of average precision mean values for cotton pests and diseases.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Cotton pest and disease categories</th>
<th valign="middle" colspan="2" align="center">mAP<sub>@0.5/%</sub>
</th>
</tr>
<tr>
<th valign="middle" align="center">Proposed Method</th>
<th valign="middle" align="center">YOLOv8s</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Nocturnal moth larvae</td>
<td valign="middle" align="center">98.6</td>
<td valign="middle" align="center">97.7%</td>
</tr>
<tr>
<td valign="middle" align="center">Cotton angular leaf spot</td>
<td valign="middle" align="center">97.1</td>
<td valign="middle" align="center">96.5</td>
</tr>
<tr>
<td valign="middle" align="center">cotton boll rot</td>
<td valign="middle" align="center">99.5</td>
<td valign="middle" align="center">99.4</td>
</tr>
<tr>
<td valign="middle" align="center">Cotton hoarfrost</td>
<td valign="middle" align="center">98.5</td>
<td valign="middle" align="center">96.1</td>
</tr>
<tr>
<td valign="middle" align="center">Health</td>
<td valign="middle" align="center">92.0</td>
<td valign="middle" align="center">90.9</td>
</tr>
<tr>
<td valign="middle" align="center">alternaria leaf spot of cotton</td>
<td valign="middle" align="center">76.5</td>
<td valign="middle" align="center">74.3</td>
</tr>
<tr>
<td valign="middle" align="center">All pests and diseases</td>
<td valign="middle" align="center">93.7</td>
<td valign="middle" align="center">92.5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From <xref ref-type="table" rid="T2"><bold>Table&#xa0;2</bold></xref>, it can be observed that the method proposed in this paper for identifying six types of cotton pests and diseases&#x2014;namely, noctuid larvae, cotton angular leaf spot, cotton boll rot, cotton powdery mildew, healthy cotton, and cotton black spot&#x2014;achieves an average precision mean (mAP@0.5) improvement compared to the original model of 0.9%, 0.6%, 0.1%, 2.4%, 1.1%, and 2.2%, respectively.</p>
<p>Among the six types of cotton pests and diseases, the average precision mean (mAP@0.5) for cotton black spot is the lowest, with only 74.3%. Analysis indicates that the blurriness of the original data images led to this subpar performance. However, with the application of our method, there is a 2.2% improvement over the original model, resulting in an overall average precision mean (mAP@0.5) of 93.7%, an increase of 1.2% compared to YOLOv8s. This demonstrates that our model&#x2019;s feature extraction capability has been enhanced for images with suboptimal quality. The results of the method proposed in this paper for identifying cotton pests and diseases are shown in <xref ref-type="fig" rid="f10"><bold>Figure&#xa0;10</bold></xref>:</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Recognition results of this paper&#x2019;s method: <bold>(A)</bold> Nocturnal moth larvae <bold>(B)</bold> Cotton angular leaf spot <bold>(C)</bold> cotton boll rot <bold>(D)</bold> Cotton hoarfrost <bold>(E)</bold> Health and <bold>(F)</bold> Alternaria leaf spot of cotton.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g010.tif"/>
</fig>
</sec>
<sec id="s3_1_3">
<label>3.1.3</label>
<title>Ablation study results</title>
<p>To validate the efficacy of the improvements made to the original algorithm by the cotton pest and disease identification method based on the CFNet-VoV-GCSP-LSKNet-YOLOv8s network structure proposed in this paper, an ablation study was designed. The original network and the network improved with various modules were tested on a test dataset. The results are shown in <xref ref-type="table" rid="T3"><bold>Table&#xa0;3</bold></xref>:</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Ablation test results.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Baseline model</th>
<th valign="middle" align="center">CFNet</th>
<th valign="middle" align="center">VoV-GCSP</th>
<th valign="middle" align="center">LSKNet</th>
<th valign="middle" align="center">XIoU</th>
<th valign="middle" align="center">P/%</th>
<th valign="middle" align="center">R/%</th>
<th valign="middle" align="center">mAP<sub>@0.5/%</sub>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Model 1</td>
<td valign="middle" align="center">YOLO v8s</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center">87.9</td>
<td valign="middle" align="center">89.7</td>
<td valign="middle" align="center">92.5</td>
</tr>
<tr>
<td valign="middle" align="center">Model 2</td>
<td valign="middle" align="center">YOLO v8s</td>
<td valign="middle" align="center"><bold>&#x221a;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center">90.6</td>
<td valign="middle" align="center">89.4</td>
<td valign="middle" align="center">93.3</td>
</tr>
<tr>
<td valign="middle" align="center">Model 3</td>
<td valign="middle" align="center">YOLO v8s</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#x221a;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center">87.4</td>
<td valign="middle" align="center">90.7</td>
<td valign="middle" align="center">92.9</td>
</tr>
<tr>
<td valign="middle" align="center">Model 4</td>
<td valign="middle" align="center">YOLO v8s</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#x221a;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center">88.9</td>
<td valign="middle" align="center">88.2</td>
<td valign="middle" align="center">92.7</td>
</tr>
<tr>
<td valign="middle" align="center">Model 5</td>
<td valign="middle" align="center">YOLO v8s</td>
<td valign="middle" align="center"><bold>&#x221a;</bold>
</td>
<td valign="middle" align="center"><bold>&#x221a;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center"><bold>&#xd7;</bold>
</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">89.8</td>
<td valign="middle" align="center">93.4</td>
</tr>
<tr>
<td valign="middle" align="center">Model 6</td>
<td valign="middle" align="center">YOLO v8s</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">89.9</td>
<td valign="middle" align="center">90.7</td>
<td valign="middle" align="center">93.7</td>
</tr>
</tbody>
</table>
</table-wrap>
<list list-type="order">
<list-item>
<p>Model 2, the first improvement, replaces all C2F modules in the backbone with CFNet modules, resulting in a 2.7% increase in Precision (P), a 0.3% increase in Recall (R), and a 0.8% increase in Mean Average Precision (mAP). This indicates that the CFNet module effectively fuses multi-scale features and improves model recognition accuracy.</p>
</list-item>
<list-item>
<p>Model 3 replaces all C2F modules in the YOLOv8s head with VoV-GCSP modules, leading to a 1% increase in R and a 0.4% increase in mAP. However, P decreased by 0.5%, but the overall performance is still better than the original model, suggesting that the neck structure composed of VoV-GCSP modules enhances the features extracted by the backbone.</p>
</list-item>
<list-item>
<p>Model 4 adds the LSKNet attention mechanism compared to YOLOv8s, resulting in a 1% increase in P and a 0.2% increase in mAP, suggesting that the LSKNet attention mechanism strengthens the model&#x2019;s ability to recognize small objects.</p>
</list-item>
<list-item>
<p>Model 5 incorporates both CFNet and VoV-GCSP modules, leading to a 1.2% increase in P, a 0.1% increase in R, and a 0.9% increase in mAP.</p>
</list-item>
<list-item>
<p>Model 6, based on the improvements in Model 5, further incorporates the LSKNet attention mechanism and replaces the loss function with XIoU. It turns out that Model 6 has the highest precision among all the models. Compared to the original model, it increases P by 2%, R by 1%, and mAP by 1.2%, demonstrating that the improved model outperforms YOLOv8s in recognition performance and effectively enhances cherry detection capabilities.</p>
</list-item>
</list>
<p>To further validate the effectiveness and practicality of the method proposed in this paper, the location loss values of CFNet-VoV-GCSP-LSKNet-YOLOv8s and YOLOv8s are shown in <xref ref-type="fig" rid="f11"><bold>Figure&#xa0;11</bold></xref> after 150 training iterations. As can be seen from <xref ref-type="fig" rid="f11"><bold>Figure&#xa0;11</bold></xref>, the convergence speed of the proposed method is faster, and its convergence performance is superior to that of the YOLOv8s model.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Comparison of positional loss values.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g011.tif"/>
</fig>
</sec>
<sec id="s3_1_4">
<label>3.1.4</label>
<title>Comparison of different models</title>
<p>To verify the effectiveness of the cotton pest and disease identification method based on the CFNet-VoV-GCSP-LSKNet-YOLOv8s model proposed in this paper, we compared it with YOLO v5s (<xref ref-type="bibr" rid="B7">Jiang et&#xa0;al., 2022</xref>), YOLOX (<xref ref-type="bibr" rid="B5">Ge et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B29">Zhang et&#xa0;al., 2022b</xref>), YOLOv7 (<xref ref-type="bibr" rid="B3">Cao et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B21">Wang C-Y. et&#xa0;al., 2023</xref>), Faster R-CNN (<xref ref-type="bibr" rid="B9">Li, 2021</xref>; <xref ref-type="bibr" rid="B17">Qiao et&#xa0;al., 2021</xref>).</p>
<p>YOLO v8s, YOLOv8n, YOLOv7-tiny, CenterNet and EfficientDet. Among them, YOLOv8s, YOLO v5s, YOLOv7, YOLOv8n, YOLOv7-tiny, CenterNet and EfficientDet are currently mainstream object detection algorithms, while YOLO X and Faster R-CNN have shown better performance in other studies. To validate the superiority of the proposed method, all model training processes maintained consistent parameter settings. The comparison results are shown in <xref ref-type="table" rid="T4"><bold>Table&#xa0;4</bold></xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Comparison of recognition effect of different models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">P/%</th>
<th valign="middle" align="center">R/%</th>
<th valign="middle" align="center">mAP<sub>@0.5/%</sub>
</th>
<th valign="middle" align="center">volume modulus/MB</th>
<th valign="middle" align="center">Detection Time/ms</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLO v5s</td>
<td valign="middle" align="center">88.5</td>
<td valign="middle" align="center">88.8</td>
<td valign="middle" align="center">91.4</td>
<td valign="middle" align="center">14.4</td>
<td valign="middle" align="center">10</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOX</td>
<td valign="middle" align="center">90.6</td>
<td valign="middle" align="center">86.7</td>
<td valign="middle" align="center">88.7</td>
<td valign="middle" align="center">34.4</td>
<td valign="middle" align="center">9.1</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO v7</td>
<td valign="middle" align="center">78.5</td>
<td valign="middle" align="center">79.1</td>
<td valign="middle" align="center">81.2</td>
<td valign="middle" align="center">74.8</td>
<td valign="middle" align="center">11.2</td>
</tr>
<tr>
<td valign="middle" align="center">Faster R-CNN</td>
<td valign="middle" align="center">60.8</td>
<td valign="middle" align="center">91.2</td>
<td valign="middle" align="center">84.7</td>
<td valign="middle" align="center">521.9</td>
<td valign="middle" align="center">9.9</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO v8n</td>
<td valign="middle" align="center">84.4</td>
<td valign="middle" align="center">85.3</td>
<td valign="middle" align="center">90.0</td>
<td valign="middle" align="center">6.2</td>
<td valign="middle" align="center">7.0</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv7-tiny</td>
<td valign="middle" align="center">87.7</td>
<td valign="middle" align="center">41.5</td>
<td valign="middle" align="center">71.9</td>
<td valign="middle" align="center">23.5</td>
<td valign="middle" align="center">8.2</td>
</tr>
<tr>
<td valign="middle" align="center">CenterNet</td>
<td valign="middle" align="center">95.8</td>
<td valign="middle" align="center">65.1</td>
<td valign="middle" align="center">88.8</td>
<td valign="middle" align="center">124</td>
<td valign="middle" align="center">10.4</td>
</tr>
<tr>
<td valign="middle" align="center">EfficientDet</td>
<td valign="middle" align="center">87.8</td>
<td valign="middle" align="center">70.7</td>
<td valign="middle" align="center">81.6</td>
<td valign="middle" align="center">25.7</td>
<td valign="middle" align="center">9.3</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO v8s</td>
<td valign="middle" align="center">87.9</td>
<td valign="middle" align="center">89.7</td>
<td valign="middle" align="center">92.5</td>
<td valign="middle" align="center">21.4</td>
<td valign="middle" align="center">7.7</td>
</tr>
<tr>
<td valign="middle" align="center">Proposed Method</td>
<td valign="middle" align="center">&gt;89.9</td>
<td valign="middle" align="center">&gt;90.7</td>
<td valign="middle" align="center">&gt;93.7</td>
<td valign="middle" align="center">&gt;23.3</td>
<td valign="middle" align="center">8.01</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As delineated in <xref ref-type="table" rid="T4"><bold>Table&#xa0;4</bold></xref>, the proposed method demonstrates marked improvements in performance metrics over existing methods. Specifically: When compared with YOLO v5s, YOLO v7, YOLOv8n, YOLOv7-tiny, EfficientDet and YOLOv8s, the precision of our model improved by 1.4%, 11.4%, 5.5%, 2.2%, 2.1% and 2% respectively. Additionally, recall rates saw increases of 1.9%, 11.6%, 5.4%, 49.2%, 20% and 1%, and the mean average precision (mAP) advanced by 2.3%, 12.5%, 3.7%, 21.8%, 12.1% and 1.2%. While our method registers a slight decline in precision relative to YOLOX, it compensates with a 4% increase in recall and a 5% boost in mAP. As for Faster R-CNN, although the recall rate was marginally lower by 0.5%, the model achieved a substantial enhancement in precision by 29.1%, along with a 9% improvement in mAP. Compared with CenterNet, although the precision is 5.9% lower, the mAP is 4.9% higher and the recall rate is 25.6% higher. In terms of computational resource consumption, our model is second only to YOLO v5s, YOLOv8n and YOLO v8s. The detection time of the improved algorithm is 8.01ms. Although slightly lower than the fastest detection speed YOLOv8n and YOLO v8s, other performance indicators of the detection algorithm are better than this model. Therefore, based on the overall detection performance indicators of the model, the algorithm in this paper has great advantages in both recognition accuracy and speed. Collectively, these results validate the effectiveness of the proposed method, positioning it as superior in object detection performance. <xref ref-type="fig" rid="f12"><bold>Figure&#xa0;12</bold></xref> offers further insights into the comparative performance of various models. While the proposed method slightly lags behind YOLO v5s in terms of convergence speed during the initial 17 iterations, it surpasses all competing models in both convergence speed and mAP following the 17th iteration.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Variation curves of mAP for different models.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1348402-g012.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>In order to verify the robustness and effectiveness of the model, the YOLOv8s original model and the Proposed Method were tested on datasets collected from the Kaggle website, which include grape and coffee disease datasets. The grape disease dataset consists of four types of diseases (Black Rot, Grape Esca, Grape Healthy, and Leaf Blight), totaling 3330 images. The coffee disease dataset consists of ten types (Coffee White Stem Borer, Citrus Mealybug, Coffee Berry Borer, Coffee Root-knot Nematode, Coffee Berry Moth, Coffee Leaf Miner, Coffee Twig Borer, Coffee Seedling Sudden Collapse Disease, Coffee Seedling Damping-off Disease), totaling 4500 images. The experimental results are shown in the table below.</p>
<p>As shown in the table above, the precision, recall rate, and mAP of Proposed Method have improved in grape disease identification, increasing by 0.1%, 0.1%, and 0.15% respectively. The precision, recall rate, and mAP of Proposed Method in identifying coffee diseases have been greatly improved, increasing by 1%, 2.8%, and 1.8% respectively. It shows that the Proposed Method has a good recognition effect on public plant disease data sets, thus proving that the model has good robustness and effectiveness. In the future, the Proposed Method can be applied in various fields.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions</title>
<p>This paper proposes a cotton pest and disease recognition method based on CFNet-VoV-GCSP-LSKNet-YOLOv8s. By replacing all C2F modules in the backbone of YOLO v8s with CFNet modules and incorporating feature fusion operations, the method effectively utilizes a significant proportion of the backbone network to fuse multi-scale features. In the head of YOLOv8s, we replaced all C2F modules with VoV-GCSP modules. This enhances the features extracted by the backbone while reducing the model&#x2019;s complexity and maintaining its accuracy. We also introduced the LSKNet attention mechanism in both the backbone and the head to improve the model&#x2019;s ability to recognize small targets. Finally, the XIoU loss function was introduced to improve the model&#x2019;s convergence performance. As shown in <xref ref-type="table" rid="T5"><bold>Table&#xa0;5</bold></xref>: experimental results show that the proposed method can effectively identify cotton pests and diseases with an average accuracy of 93.7%, demonstrating its effectiveness.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Comparison of results for different diseases.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Data</th>
<th valign="top" align="center">Model</th>
<th valign="top" align="center">P/%</th>
<th valign="top" align="center">R/%</th>
<th valign="top" align="center">mAP<sub>@0.5%</sub>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="2" align="center">Grape diseases</td>
<td valign="top" align="center">YOLOv8s</td>
<td valign="top" align="center">99.30</td>
<td valign="top" align="center">99.60</td>
<td valign="top" align="center">99.25</td>
</tr>
<tr>
<td valign="top" align="center">Proposed Method</td>
<td valign="top" align="center">99.40</td>
<td valign="top" align="center">99.70</td>
<td valign="top" align="center">99.40</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="center">Coffee diseases</td>
<td valign="top" align="center">YOLOv8s</td>
<td valign="top" align="center">80.20</td>
<td valign="top" align="center">76.30</td>
<td valign="top" align="center">81.80</td>
</tr>
<tr>
<td valign="top" align="center">Proposed Method</td>
<td valign="top" align="center">81.20</td>
<td valign="top" align="center">79.10</td>
<td valign="top" align="center">83.60</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Compared with YOLO v5s, YOLOX, YOLO v7, Faster R-CNN, YOLOv8n, YOLOv7-tiny, CenterNet, EfficientDet and YOLOv8s, the average accuracy improved by 2.5%, 5%, 12.5%, 9%, 3.7%, 21.8%, 4.9%, 12.1% and 1.2% respectively, indicating that the method proposed in this paper performs better in recognizing cotton pests and diseases. In addition, the model in this article was applied to grape and coffee diseases, which greatly improved the disease identification rate, indicating that the model has good robustness and effectiveness.</p>
<p>While CFNet-VoV-GCSP-LSKNet-YOLOv8s demonstrates considerable prowess in numerous domains, there is still potential for advancement in its efficacy in detecting cotton leaf spot disease. Looking forward, our ambition is to enrich YOLOv8s&#x2019;s backbone network with a sophisticated multi-channel scale attention mechanism, aimed at enhancing the precision in capturing the characteristics of plant diseases. Concurrently, by refining the final prediction bounding box optimization and the Adam optimizer within YOLOv8s, we aspire to elevate the model&#x2019;s recognition proficiency. Moreover, in tandem with real-world application demands, our objective includes the development of mobile applications. This endeavor is geared towards translating our research findings into pragmatic tools, offering robust and practical solutions in the realms of agriculture and plant protection, thereby facilitating the deployment of this technology in real-world scenarios.</p>
</sec>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <uri xlink:href="https://www.kaggle.com/datasets/paridhijain02122001/cotton-crop-disease-detection">https://www.kaggle.com/datasets/paridhijain02122001/cotton-crop-disease-detection</uri> and <uri xlink:href="https://www.kaggle.com/datasets/saeedazfar/customized-cotton-disease-dataset">https://www.kaggle.com/datasets/saeedazfar/customized-cotton-disease-dataset</uri>.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>RL: Conceptualization, Methodology, Resources, Validation, Writing &#x2013; original draft. YH: Data curation, Methodology, Software, Validation, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. YL: Data curation, Resources, Validation, Writing &#x2013; review &amp; editing. WQ: Data curation, Software, Writing &#x2013; review &amp; editing. AA: Data curation, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. RJ: Data curation, Methodology, Writing &#x2013; review &amp; editing. SL: Methodology, Writing &#x2013; review &amp; editing. YW: Data curation, Writing &#x2013; review &amp; editing. XS: Software, Writing &#x2013; review &amp; editing. JY: Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The authors would like to express their sincere gratitude for the financial support provided by the Major Project of Yunnan Science and Technology, under Project No. 202302AE09002003, entitled &#x201c;Research on the Integration of Key Technologies in Smart Agriculture.&#x201d;</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author SX was employed by Jilin Haicheng Technology Co.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abbas</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Iqbal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ahmad</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Toxicity of selective insecticides against sap sucking insect pests of cotton (Gossypium hirsutum)</article-title>. <source>Pure Appl. Biol.</source> <volume>11</volume> (<issue>1</issue>), <fpage>72</fpage>&#x2013;<lpage>78</lpage>. doi: <pub-id pub-id-type="doi">10.19045/bspab.2022-110008</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Caldeira</surname> <given-names>R. F.</given-names>
</name>
<name>
<surname>Santiago</surname> <given-names>W. E.</given-names>
</name>
<name>
<surname>Teruel</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Identification of cotton leaf?lesions using deep learning techniques</article-title>. <source>Sensors</source> <volume>21</volume> (<issue>9</issue>), <fpage>3169</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s21093169</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The semantic segmentation of standing tree images based on the yolo V7 deep learning algorithm</article-title>. <source>Electronics</source> <volume>12</volume> (<issue>4</issue>), <fpage>929</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/electronics12040929</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chohan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Perveen</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Abid</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Tahir</surname> <given-names>M. N.</given-names>
</name>
<name>
<surname>Sajid</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Cotton diseases and their management</article-title>,&#x201d; in?<source>Cotton?production uses: agronomy, crop protection, postharvest technologies</source>, (<publisher-name>Springer</publisher-name>), <fpage>239</fpage>&#x2013;<lpage>270</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-981-15-1472-2_13</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ge</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Yolox: Exceeding yolo series in 2021</article-title>. <source>arXiv preprint</source> <publisher-name>arXiv:2107.08430</publisher-name>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2107.08430</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jajja</surname> <given-names>A. I.</given-names>
</name>
<name>
<surname>Abbas</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Khattak</surname> <given-names>H. A.</given-names>
</name>
<name>
<surname>Niedba&#x142;a</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Khalid</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Rauf</surname> <given-names>H. T.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Compact convolutional transformer (CCT)-Based approach for whitefly attack detection in cotton crops</article-title>. <source>Agriculture</source> <volume>12</volume> (<issue>10</issue>), <fpage>1529</fpage>. doi: <pub-id pub-id-type="doi">10.3390/agriculture12101529</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Ergu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A Review of Yolo algorithm developments</article-title>. <source>Proc. Comput. Sci.</source> <volume>199</volume>, <fpage>1066</fpage>&#x2013;<lpage>1073</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2022.01.135</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Slim-neck by GSConv: A better design paradigm of detector architectures for autonomous vehicles</source> (<publisher-name>arXiv</publisher-name>), <fpage>02424</fpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Analysis of object detection performance based on faster R-CNN</article-title>. <source>J. Physics: Conf. Ser.</source> <volume>1827</volume> (<issue>1</issue>), <fpage>012085</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1088/1742-6596/1827/1/012085</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hou</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Large selective kernel network for remote sensing object detection</source> (<publisher-name>arXiv</publisher-name>), <fpage>09030</fpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname> <given-names>X. H. Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Few-shot cotton leaf spots disease classification based on metric learning</article-title>. <source>Plant Methods</source> <volume>17</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s13007-021-00813-7</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Tomato diseases and pests detection based on improved Yolo V3 convolutional neural network</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>, <elocation-id>898</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2020.00898</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Small unopened cotton boll counting by detection with MRF-YOLO in the wild</article-title>. <source>Comput. Electron. Agric.</source> <volume>204</volume>, <fpage>107576</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2022.107576</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meng</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Transforming unmanned pineapple picking with spatio-temporal convolutional neural networks</article-title>. <source>Comput. Electron. Agric.</source> <volume>214</volume>, <fpage>108298</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2023.108298</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nannan</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Tiecheng</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Cotton leaf pest and disease recognition method under natural environment based on CBAM-YOLO v7[J/OL]</article-title>. <source>J.?Agric. Machinery</source> <volume>2020</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>.
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Patil</surname> <given-names>B. V.</given-names>
</name>
<name>
<surname>Patil</surname> <given-names>P. S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Computational method for Cotton Plant disease detection of crop management using deep learning and internet of things platforms</article-title>,&#x201d; in <source>Evolutionary computing and mobile sustainable networks: proceedings of ICECMSN 2020</source> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer</publisher-name>), pp. <fpage>875</fpage>&#x2013;<lpage>885</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Qiao</surname> <given-names>L. M.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>DeFRCN: decoupled faster R-CNN for few-shot object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision</conf-name> (<publisher-name>Electr Network</publisher-name>), pp. <fpage>8681</fpage>&#x2013;<lpage>8690</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Niu</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>PLANET: improved convolutional neural networks with image enhancement for image classification</article-title>. <source>Math. Problems Eng.</source> <volume>2020</volume>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi: <pub-id pub-id-type="doi">10.1155/2020/5892312</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Su</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Cotton disease detection based on convNeXt and attention mechanisms</article-title>. <source>IEEE J. Radio Frequency Identification</source> <volume>6</volume>, <fpage>805</fpage>&#x2013;<lpage>809</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JRFID.2022.3206841</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Terven</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cordova-Esparza</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <source>A comprehensive review of YOLO: From YOLOv1 to YOLOv8 and beyond</source>, arXiv preprint arXiv:.00501.</citation>
</ref>
<ref id="B21">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A performance analysis of a litchi picking robot system for actively removing obstructions, using an artificial intelligence algorithm</article-title>. <source>Agronomy</source> <volume>13</volume> (<issue>11</issue>), <fpage>2795</fpage>. doi: <pub-id pub-id-type="doi">10.3390/agronomy13112795</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>An</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>UAV-YOLOv8: A small-object-detection model based on improved YOLOv8 for UAV aerial photography scenarios</article-title>. <source>Sensors</source> <volume>23</volume> (<issue>16</issue>), <fpage>7190</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s23167190</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Tea-YOLOv8s: A tea bud detection model based on deep learning and computer vision</article-title>. <source>Sensors</source> <volume>23</volume> (<issue>14</issue>), <fpage>6576</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s23146576</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xue</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>YOLO-tea: A tea disease detection model improved by YOLOv5</article-title>. <source>Forests</source> <volume>14</volume> (<issue>2</issue>), <fpage>415</fpage>. doi: <pub-id pub-id-type="doi">10.3390/f14020415</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Path planning for mobile robots in unstructured orchard environments: An improved kinematically constrained bi-directional RRT approach</article-title>. <source>Comput. Electron. Agric.</source> <volume>215</volume>, <fpage>108453</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2023.108453</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Cfnet: Cascade fusion network for dense prediction</source> (<publisher-name>arXiv</publisher-name>), <fpage>2302.06052</fpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>a). <article-title>Accurate cotton diseases and pests detection in complex background based on an improved YOLOX model</article-title>. <source>Comput. Electron. Agric.</source> <volume>203</volume>, <fpage>107484</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2022.107484</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>He</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>He</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>b). <article-title>Complete and accurate holly fruits counting using YOLOX object detection</article-title>. <source>Comput. Electron. Agric.</source> <volume>198</volume>, <fpage>107062</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2022.107062</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>CURI-YOLOv7: A lightweight YOLOv7tiny target detector for citrus trees from UAV remote sensing imagery based on embedded device</article-title>. <source>Remote Sens.</source> <volume>15</volume> (<issue>19</issue>), <fpage>4647</fpage>. doi: <pub-id pub-id-type="doi">10.3390/rs15194647</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>
