<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2024.1406593</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A lightweight Color-changing melon ripeness detection algorithm based on model pruning and knowledge distillation: leveraging dilated residual and multi-screening path aggregation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Guojun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2697998"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Hou</surname>
<given-names>Yongjie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Haozhen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cao</surname>
<given-names>Lei</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yuan</surname>
<given-names>Jianqiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Qingdao Institute of Software, College of Computer Science and Technology, China University of Petroleum (East China)</institution>, <addr-line>Qingdao</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Faculty of Light Industry, Qilu University of Technology</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>State Key Laboratory of Biobased Material and Green Papermaking, Shandong Academy of Sciences</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Christine Dewi, Satya Wacana Christian University, Indonesia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Peter Ardhianto, Soegijapranata Catholic University, Indonesia</p>
<p>Radius Tanone, Chaoyang University of Technology, Taiwan</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yongjie Hou, <email xlink:href="mailto:houyongjie@s.upc.edu.cn">houyongjie@s.upc.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1406593</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Chen, Hou, Chen, Cao and Yuan</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Chen, Hou, Chen, Cao and Yuan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Color-changing melons are a kind of cucurbit plant that combines ornamental and food. With the aim of increasing the efficiency of harvesting Color-changing melon fruits while reducing the deployment cost of detection models on agricultural equipment, this study presents an improved YOLOv8s network approach that uses model pruning and knowledge distillation techniques. The method first merges Dilated Wise Residual (DWR) and Dilated Reparam Block (DRB) to reconstruct the C2f module in the Backbone for better feature fusion. Next, we designed a multilevel scale fusion feature pyramid network (HS-PAN) to enrich semantic information and strengthen localization information to enhance the detection of Color-changing melon fruits with different maturity levels. Finally, we used Layer-Adaptive Sparsity Pruning and Block-Correlation Knowledge Distillation to simplify the model and recover its accuracy. In the Color-changing melon images dataset, the mAP0.5 of the improved model reaches 96.1%, the detection speed is 9.1% faster than YOLOv8s, the number of Params is reduced from 6.47M to 1.14M, the number of computed FLOPs is reduced from 22.8GFLOPs to 7.5GFLOPs. The model&#x2019;s size has also decreased from 12.64MB to 2.47MB, and the performance of the improved YOLOv8 is significantly more outstanding than other lightweight networks. The experimental results verify the effectiveness of the proposed method in complex scenarios, which provides a reference basis and technical support for the subsequent automatic picking of Color-changing melons.</p>
</abstract>
<kwd-group>
<kwd>Color-changing melon</kwd>
<kwd>multi-scale feature fusion</kwd>
<kwd>model pruning</kwd>
<kwd>knowledge distillation</kwd>
<kwd>YOLOv8s</kwd>
</kwd-group>
<counts>
<fig-count count="14"/>
<table-count count="3"/>
<equation-count count="16"/>
<ref-count count="46"/>
<page-count count="18"/>
<word-count count="9226"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Technical Advances in Plant Science</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Color-changing melon, a kind of fruit that turns from green to red on the surface of its skin when it matures, belongs to the Cucurbitaceae family of vines, and the fruit is thick in the middle and thin at both ends, resembling a mouse, so it is also known as the mouse melon. It is suitable for planting in the garden, both ornamental and edible. When cultivated on the plantation, workers will plant the seedlings in the hanging soil, and when the plant grows up, the vine climbs all over the shelves. The fruit naturally hangs down with a reddish color for a good ornamental appearance. At the same time, Color-changing melons have a high yield, and a single plant can get about 200 fruits in its lifetime. Therefore, after some ornamental fruits are left behind, most of the remaining immature fruits are picked and used in stir-fries or soups for a refreshing flavor. A Color-changing melon plant can produce fruit for up to five consecutive months. Due to the varying maturity periods of the fruits, in the current production environment, the immature fruits are mainly picked by hand. Fruits picked too early have a rugged quality and poor flavor, while fruits picked too late lose their food value and affect profitability (<xref ref-type="bibr" rid="B4">Camposeo et&#xa0;al., 2013</xref>). If you rely only on workers, you need to pick several times, which is too time-consuming and inefficient (<xref ref-type="bibr" rid="B42">Yang et&#xa0;al., 2023</xref>). Meanwhile, the fruits are all growing on 3-meter-high shelves, and picking operations that do not meet safety norms increase the risk of worker injury. In response to these problems, we believe robotic arms (<xref ref-type="bibr" rid="B19">Kang et&#xa0;al., 2020</xref>) can be developed to automatically pick fruits that meet standards. This can alleviate the problem of labor shortage in agricultural production (<xref ref-type="bibr" rid="B8">Clark et&#xa0;al., 2018</xref>) and, at the same time, ensure the quality of picking, improve productivity, and ensure the safety of workers. However, there are still some difficulties in robotic picking technology, and the critical step is the localization and judgment of the fruit. The study of how to realize accurate target detection is a prerequisite for automatic picking work.</p>
<p>In the early field of target detection, researchers designed detection algorithms based on the fruit&#x2019;s color, shape, and texture. However, for fruits whose fruit color is similar to that of leaves, such as cucumbers, it is impossible to distinguish the fruit from the background by relying on shape alone. Therefore, researchers usually use morphology in conjunction with other methods in the process of designing algorithms. For example, Dorj et&#xa0;al (<xref ref-type="bibr" rid="B10">Dorj et&#xa0;al., 2017</xref>). designed algorithms to detect citrus based on color and shape features. However, traditional algorithms are only designed for a specific scene. If the interference of environmental factors such as light changes is considered (<xref ref-type="bibr" rid="B44">Zhang et&#xa0;al., 2022</xref>), the detection effect on the target will be significantly reduced. Traditional machine learning algorithms have some improvements in detecting fruits. However, they still have similar problems: they often need to limit the types of features to compress the feature space (<xref ref-type="bibr" rid="B5">Chaudhari and Waghmare, 2022</xref>), they cannot learn high-dimensional features directly, and they are not robust and generalized enough to face a variety of complex scenes.</p>
<p>As science continues to develop, Convolutional Neural Networks (CNNs) have overcome the limitations of traditional machine learning and demonstrated excellent performance (<xref ref-type="bibr" rid="B20">Krizhevsky et&#xa0;al., 2017</xref>). CNN-based machine vision has been increasingly widely used in agriculture (<xref ref-type="bibr" rid="B18">Kamilaris and Prenafeta-Bold&#xfa;, 2018</xref>), and the resulting deep-learning networks are continuously penetrating the field of Computer Vision. With the structure of CNNs as the Backbone, the model extracts rich feature information and dramatically improves the accuracy of detection, while the high-dimensional features processed by multi-layer convolution further enhance the generalization of different application scenarios. Classical target detection algorithms consist of a classification process and a localization process, and these algorithms can be classified into two-stage detection algorithms and one-stage detection algorithms based on whether they produce candidate regions. The R-CNN family of networks are representative algorithms for two-stage detection, which first generate candidate regions and then perform the target classification task and the target localization task separately, for instance, Faster R-CNN (<xref ref-type="bibr" rid="B34">Ren et&#xa0;al., 2015</xref>) and Mask R-CNN (<xref ref-type="bibr" rid="B14">He et&#xa0;al., 2017</xref>). Mu et&#xa0;al (<xref ref-type="bibr" rid="B30">Mu et&#xa0;al., 2020</xref>). used Faster R-CNN incorporating transfer learning and achieved a mAP of 87.83% on a homemade immature tomato dataset. Jin et&#xa0;al (<xref ref-type="bibr" rid="B36">Suddapalli and Shyam, 2021</xref>). used Mask R-CNN to segment diseased portions of vegetables and fruits, which was used in place of the manual screening process.</p>
<p>In contrast, one-stage detection algorithms have faster detection speeds and such algorithms are represented by SSD (<xref ref-type="bibr" rid="B27">Liu et&#xa0;al., 2016</xref>) and the YOLO family (<xref ref-type="bibr" rid="B33">Redmon et&#xa0;al., 2016</xref>). The YOLO family has iterated many versions through continuous development (<xref ref-type="bibr" rid="B38">Terven et&#xa0;al., 2023</xref>), with progressively improved extensibility and generalization, and is now widely used in detecting fruits and diseases in agriculture (<xref ref-type="bibr" rid="B35">Shi et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B37">Suo et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B31">Nan et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B46">Zhu et&#xa0;al., 2024</xref>). Liang et&#xa0;al (<xref ref-type="bibr" rid="B25">Liang et&#xa0;al., 2020</xref>). combined YOLOv3 and UNet to detect lychee under nighttime conditions. YOLOv3 suffers from the problem of a relatively complex model structure. Therefore, subsequent research has also focused on lightweight target detection algorithms. Li et&#xa0;al (<xref ref-type="bibr" rid="B23">Li et&#xa0;al., 2021</xref>). modified the YOLOv4-Tiny model to design a detection algorithm for corn kernel breakage during harvesting and provide parameters for the combined harvester while working. Zeng et&#xa0;al (<xref ref-type="bibr" rid="B43">Zeng et&#xa0;al., 2023</xref>). used the MobileNetv3 network to replace Backbone in YOLOv5 while optimizing the training hyperparameters. They constructed a lightweight model successfully deployed to cell phones to detect tomatoes&#x2019; maturity. Nouaze et&#xa0;al (<xref ref-type="bibr" rid="B32">Nouaze and Sikati, 2023</xref>). introduced the FEature architecture in YOLOv7, which was used to combine various pieces of information in the feature space and increase the model&#x2019;s recognition accuracy for both healthy and diseased apples, and its recognition accuracy with a mAP of 89.30%.</p>
<p>Although the above-improved algorithms have made progress in model lightweight, they only pursue the simplification of network structure when they face complex real-world scenarios, such as backlighting, overlapping fruits, dense fruits, and fruits being occluded by other objects, many of the target detection algorithms that have been lightweight are limited by the small number of parameters and computation, their robustness is not ideal, and they often miss and misdetect, which makes it difficult to cope with the detection in complex scenes. Therefore, in response to the challenge, most lightweight models are less robust when facing complex scenes. In contrast, high-accuracy models suffer from a more complex network structure; this paper constructs an improved model based on YOLOv8s as well as a series of subsequent processing of the model, which can be used for the real-time detection task of picking robots and low-cost edge devices in complex natural environments under the premise of guaranteeing the detection accuracy.</p>
<p>These combined algorithms can effectively improve the shortages of sizeable computational cost and excessive memory occupation during the model deployment while maintaining a high accuracy rate, providing technical support for the subsequent automatic harvesting.</p>
<p>The main contributions and innovations of this study are summarized as follows:</p>
<list list-type="simple">
<list-item>
<p>(1) We created a dataset of Color-changing melon figures using manual annotation, and the fruits in various real scenarios were considered in the shooting process.</p>
</list-item>
<list-item>
<p>(2) In order to improve the accuracy of target detection in complex scenes while maintaining the lightweight structure of the model, we first designed the DWR-DRB module to replace the Bottleneck in the C2f module to increase the receptive field without increasing the depth of the network, to enrich the multi-scale contextual information extracted by Backbone. Then, we constructed the HS-PAN architecture, which adopts multi-level feature fusion to aggregate multi-scale features and can effectively focus on the fruits that are interfered with by background factors.</p>
</list-item>
<list-item>
<p>(3) After the model was trained, we used layer adaptive sparsity pruning on the model, and the pruned model computed only one-third of the original FLOPs. Then, using the improved model trained in advance as the teacher model, the pruned model is distilled using block correlation knowledge distillation, and the student model does not increase the network complexity. At the same time, the recognition performance is further improved, which helps the model to be deployed on mobile terminals or embedded devices with limited resources.</p>
</list-item>
<list-item>
<p>(4) We conducted a series of comparison experiments related to Color-changing melon dataset detection. We first conducted a comparison of the model recognition effect before and after improvement, then designed an ablation experiment, next compared the number of each channel before and after model pruning and the effect of different scales of teacher models on the distillation effect, and finally compared our improved algorithm with other lightweight algorithms to demonstrate the difference in performance between different algorithms.</p>
</list-item>
</list>
<p>The rest of the paper is organized as follows. Part II discusses the processing flow of the Color-changing melon dataset and the improved YOLOv8s model and also describes the model pruning and knowledge distillation methods used. Part III explains the experimental setup and evaluation metrics and discusses the results of the various types of comparisons. Part IV summarizes.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Data acquisition</title>
<p>The Color-changing melons dataset utilized in our research was collected from a vegetable science and technology park in Shouguang City, Shandong Province (36&#xb0;51&#x2032;N, 118.49&#x2032;E), and photographed during July 2023, every day from 10:00 a.m. to 4:00 p.m. All images were obtained using the Sony IMX 866 rear camera of the Vivo X80 smartphone under natural lighting conditions. The shooting distance ranged from 0.8 meters to 1.2 meters. The images consider variations in factors such as shooting angle, lighting, and fruit overlap. After filtering out low-quality images, such as overexposure and severe blurring, 1240 images were finally obtained and archived in a JPG file type of <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mn>4032</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3024</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> pixels. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> displays the sample data obtained from various shooting scenarios.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Images captured under different scenes.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g001.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Data labeling</title>
<p>In the task of detecting the maturity of Color-changing melons, the maturity of Color-changing melons was classified into green immature, orange semi-mature, and red mature stages based on the color of the fruit surface in accordance with agricultural harvesting requirements. Some of the green immature stages have fine stripes present on the surface of the fruit. When the surface of the fruit fades from green to orange starting from the top, this enters the semi-mature stage. The immature stage is reached when the color of the fruit surface gradually deepens until it turns completely red. During actual harvesting, a small number of semi-mature and mature fruits are used for ornamental purposes as well as seed reserves, while most of the immature fruits are picked for consumption. The use of algorithms to obtain information on the maturity of the fruit helps to provide a basis for judgment of the picking work of the robot. Without affecting the recognition accuracy, we set the images in the Color-changing melons dataset uniformly at <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mn>640</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>640</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> pixels and randomly divided them into the training set, validation set, and test set according to the ratio of <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:mn>8</mml:mn>
<mml:mo>:</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The images of the three maturity levels of fruits are uniformly distributed in each set without intersecting each other. There are 992 images in the training set and 124 in the validation and test sets, respectively. Then, all the images are labeled using LabelImg, and the labels are saved in txt format and converted to xml format for easy training and testing.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Data augmentation</title>
<p>To improve the trained network&#x2019;s effectiveness and enhance the model&#x2019;s robustness, data enhancement methods are used to increase the number of images in the training part to avoid overfitting. With the help of the Augmentor tool, we performed operations such as flipping, brightness adjustment, warping distortion, and adding noise to the images, 150 images were obtained for each enhancement method, and finally, the training set was expanded to 2142 images. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> provides an example of each data augmentation technique. It is worth noting that in cases where it is difficult to obtain a large number of labeled fruit figures, in addition to the offline data augmentation approach used in the paper, it is good to consider using FSL (Few-Shot Learning) or Meta-learning to help the model improve its generalization ability. These two approaches provide practical tools for dealing with data scarcity in resource-constrained environments. Meta-learning learns from a few crucial fruit samples and thus adapts quickly to different fruit maturity stages. Further, it enhances adaptation to new tasks by learning multiple related tasks and constructing similar sets between different tasks. Few shot learning is a learning strategy to improve the model&#x2019;s ability to generalize to new tasks with fewer supervised samples, and it usually utilizes prior knowledge to simplify the sample features. In practical agricultural applications, these two methods can be combined to train the base model through meta-learning first and then use few shot learning to fine-tune the model and improve its generalization in the case of limited samples to apply the target detection technology more widely to agricultural automated picking systems.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Examples of various data augmentation techniques.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g002.tif"/>
</fig>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>YOLOv8</title>
<p>Until 2016, the R-CNN family of algorithms dominated the field of target detection. After introducing YOLOv1, target detection algorithms have been differentiated into single-stage and two-stage. YOLO is characterized by abandoning the generation of candidate frames and adopting a direct regression approach for object classification and prediction. This dramatically simplifies the network structure and is nearly ten times faster than the detection speed of Faster R-CNN. As the YOLO family continues to grow, the current version of the YOLO framework has absorbed the advantages of the previous version. It is constantly innovating itself, with a broader range of applications in agriculture.</p>
<p>YOLOv8 is one of the latest YOLO architecture detectors, which inherits many of the advantages of real-time target detectors, including lightweight network architecture and powerful feature extraction capabilities with faster detection speed and higher detection accuracy. The Backbone part of YOLOv8 uses the CSPDarkNet network (<xref ref-type="bibr" rid="B3">Bochkovskiy et&#xa0;al., 2020</xref>), which applies a cross-stage hierarchical structure to the feature map merging, improving the accuracy and reducing the whole network&#x2019;s computational complexity. YOLOv8 also borrows the ELAN structure from YOLOv7 (<xref ref-type="bibr" rid="B39">Wang et&#xa0;al., 2023</xref>) and designs the C2f module, which enriches the extracted feature information. YOLOv8 uses the CIoU (<xref ref-type="bibr" rid="B45">Zheng et&#xa0;al., 2021</xref>) to determine the IoU between prediction and ground-truth frames. In addition to that, its detection header separates the classification process and localization process, introduces Distributed Focus Loss (DFL) (<xref ref-type="bibr" rid="B24">Li et&#xa0;al., 2020</xref>), and also adopts the idea of Anchor Free, which eliminates the need for predefined anchors, making it more flexible and efficient compared to previous YOLO models. YOLOv8 provides models across various scales, including nano (n), small (s), medium (m), large (l), and extra-large (x). </p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Improved YOLOv8s model</title>
<p>In this study, we considered the balance between computational cost and detection accuracy and chose YOLOv8s as the basic model. First, we designed the DWR-DRB module, which replaces the bottleneck of the original C2f module in Backbone to enhance the sensory field and constructed a new module, which we named C2f_DWR_DRB. Then, we constructed the HS-PAN architecture, which uses a bottom-up feature module to enhance the localization information. At the same time, it is combined with other layers in the Neck section to generate a more robust feature representation. The light blue background shows the central improvement part, and in the following two subsections, we describe the proposed method in detail. Our improved YOLOv8s model is shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Improvement of YOLOv8 structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g003.tif"/>
</fig>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Dilation-wise Residual-Dilated Reparam Block</title>
<p>In a dilated convolutional layer, a dilated convolutional layer utilizing a compact kernel is equated to a non-dilated (i.e., <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) convolutional layer with a larger, sparser kernel, provided that disregarding specific input pixels is analogous to interspersing additional zeroes within the convolutional kernel. The original convolution kernel <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mtext>W</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211b;</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> becomes <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msup>
<mml:mtext>W</mml:mtext>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211b;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> after insertion of zeros, a process that can be realized by the transposed convolution of <xref ref-type="disp-formula" rid="eq1">Equation (1)</xref> and the unitary kernel <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mtext>I</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211b;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Based on this equivalent conversion, DRB (Dilated Reparam Block) was proposed by Ding et&#xa0;al (<xref ref-type="bibr" rid="B9">Ding et&#xa0;al., 2023</xref>). in UniRepLKNet, which applies a solitary, unenlarged small kernel alongside several dilated small kernel layers to enhance a convolutional layer with a non-dilated large kernel.</p>
<p>By reparameterizing multiple blocks consisting of small kernel convolution layers with different dilated rates to be equivalently converted into a solitary large kernel convolution layer with a larger sparse kernel, DRB improves the detection network&#x2019;s performance to extract spatial information while maintaining the number of learned Params and computational efficiency. This design innovation provides the convolutional network with a wider receptive field without increasing the depth of the model.</p>
<p>The essence of the DWR (Dilation-wise Residual) (<xref ref-type="bibr" rid="B41">Wei et&#xa0;al., 2022</xref>) module is a two-stage method for gathering information contextually across various scales, structured around a residual framework to capture nuanced details. The multi-scale sensory wild-formed feature maps are then fused, which reduces the difficulty of acquiring information. The first step is to generate relevant residual features based on the input features. The combination of <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolutional layers, BN layers, and ReLU layers generates many feature maps of different sizes as the material for the second step of morphological filtering. The second step is to perform morphological filtering on region features of different sizes. Initially, the region feature maps are segmented into several clusters, and then different groups are convolved in different ways.</p>
<p>We notice the similarity between the deep dilated convolution in the original DWR module and DRB. They both obtain a larger receptive field by improving the dilated convolution. Therefore, we utilize the method of reparameterizing and enhancing the non-dilated large kernel convolution layer in DRB to design the DWR-DRB module to replace the Bottleneck in C2f, which is utilized to gather information from various scales more efficiently, streamlining the process of contextual understanding. Specifically, we replace the deep convolution with dilated convolution of 3 in the second branch of the original DWR module with a DRB with a convolution kernel size of <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the deep convolution with dilated convolution of 5 in the third branch with a DRB with a convolution kernel size of <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and the <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> deep convolution in the first branch with dilated convolution of 0, and thus remains unchanged. In addition, the initial branch&#x2019;s output channel was expanded to double the capacity compared to the subsequent branches due to the fact that more extensive spatial spanning connections require the help of more minor spanning connections.</p>
<p>After plotting multi-scale contextual data, various results are consolidated to link all feature mappings. Features are then merged by batch normalization and point-by-point convolution and appended to the input feature map to build a more robust and holistic expression of the features. Schematic representation of the DWR module structure. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> illustrates the three-branch DWR-DRB module of the high-level network structure. Conv denotes convolution, DConv denotes deep convolution, and c denotes the number of channels in a feature map.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>DWR-DRB (Dilation-wise Residual-Dilated Reparam Block) structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g004.tif"/>
</fig>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>High-level Screening-path Aggregation Networks</title>
<p>The color of the surface of immature fruits is close to the color of the surrounding leaves and canes, and coupled with the disruption caused by elements like fluctuating illumination and occlusion, the difference between Color-changing melons and the complex background becomes significant. To solve this problem, we refer to the multilevel feature fusion approach of HS-FPN (High-level Screening-feature Pyramid Networks) (<xref ref-type="bibr" rid="B7">Chen et&#xa0;al., 2024</xref>) and design the HS-PAN (High-level Screening-path Aggregation Networks) architecture for fusing multi-scale feature information to reduce the interference of complex backgrounds, thus improving the accuracy of fruit detection.</p>
<p>The structure of HS-PAN is shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. It consists of two sub-modules:(1) Feature processing module. (2) Feature fusion module. First, HS-FPN sieves through the feature maps derived from the Backbone at varying scales, subsequently amalgamating the information from upper and lower levels in the filtered feature maps using the Selective Feature Fusion (SFF) mechanism. Subsequently, in order to solve the drawback of FPN (<xref ref-type="bibr" rid="B26">Lin et&#xa0;al., 2017</xref>) in dealing with the ambiguity of high-level information, we constructed a bidirectional multilevel feature fusion PAN (<xref ref-type="bibr" rid="B29">Liu et&#xa0;al., 2018</xref>), i.e., HS-PAN, which adds a bottom-up feature fusion module, takes the low-level information as one of the input parts during feature fusion, and strengthens the localization information by taking advantage of the fact that the low-level features are more accurate for the target localization, which is the reason why we used the conventional C2f as a feature fusion mechanism in the beginning of the This is the reason why we use the regular C2f module when extracting features. This multilevel fusion approach has rich and comprehensive semantic information, which helps to obtain more detailed features in Color-changing melon images, thus enhancing the detection ability of the model.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>HS-PAN (High-level Screening-path Aggregation Networks) structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g005.tif"/>
</fig>
<p>We first introduce the Channel Attention (CA) and Dimensional Matching (DM) modules in the feature processing module. The CA module initially conducts global max and average pooling on the provided feature maps. This dual pooling strategy captures both the average and the critical features present. These are designed to filter out redundant information, compress features, and reduce the number of parameters. Combining the two pooling methods helps extract the critical information in each channel while ensuring minimal information loss. Next, the generated features are aggregated, and the Sigmoid function is employed to calculate the channel-wise weights in the network, which ultimately yields the weights for all channels. Subsequently, the weight information is multiplied by the feature maps of the corresponding scales to generate the filtered feature maps. The DM module adopts the point-by-point convolution method to match the feature maps of different scales and different numbers of channels before feature fusion and, at the same time, reduces the number of channels in each layer of the feature maps to 256. The SFF module, which is one of the core components of the HS-PAN, uses the high-level features as the filters to refine the low-level important information in the features to fuse multi-scale features more efficiently.</p>
<p>As illustrated in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, given a high-level feature <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and a low-level feature <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im14">
<mml:mi>C</mml:mi>
</mml:math>
</inline-formula> denotes the channel count, <inline-formula>
<mml:math display="inline" id="im15">
<mml:mi>H</mml:mi>
</mml:math>
</inline-formula> stands for the feature map&#x2019;s height, and <inline-formula>
<mml:math display="inline" id="im16">
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula> stands for the feature map&#x2019;s width. The high-level features are first dilated convolution, which is applied by a transposed convolution (T-Conv) with a 2-step and a <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution to obtain the feature <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:msub>
<mml:mo>'</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Then, the high-level features are up-sampled or down-sampled using bilinear interpolation to align the dimensions of high-level and low-level features to obtain the feature <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Next, the CA module is utilized to unify the dimensionality of the attention weights generated from converting the high-level features and filtering the low-level features. Finally, the high-level features are fused with the filtered low-level features to obtain a more comprehensive feature <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. <xref ref-type="disp-formula" rid="eq2">Equations (2</xref>, <xref ref-type="disp-formula" rid="eq3">3)</xref> illustrate the process of feature fusion, where BL (<xref ref-type="bibr" rid="B6">Chen et&#xa0;al., 2018</xref>) is a multi-scale feature representation method.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>SFF (Selective Feature Fusion) module structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g006.tif"/>
</fig>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>&#x3c5;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_8">
<label>2.8</label>
<title>Layer-Adaptive Sparsity Pruning</title>
<p>Model pruning productively decreases the number of model parameters and FLOPs (<xref ref-type="bibr" rid="B22">Lei et&#xa0;al., 2017</xref>). In neural networks, some parameters with relatively small weights take up a large amount of computational resources, but these redundant parameters have little effect on the results of model inference. By removing these parameters with smaller weights, the model can be compressed with little loss of accuracy, thus reducing memory consumption and alleviating computational load (<xref ref-type="bibr" rid="B28">Liu et&#xa0;al., 2017</xref>). Previous studies have found (<xref ref-type="bibr" rid="B11">Gale et&#xa0;al., 2019</xref>, <xref ref-type="bibr" rid="B12">Gale et&#xa0;al., 2020</xref>) that if the layered sparsity is chosen for a neural network, then a simple magnitude-based pruning (MP) can be a suitable balance between the model&#x2019;s performance and lightweight. However, there is no clear solution to choosing the hierarchical sparsity.</p>
<p>The model pruning method used in our research is the magnitude-based Layer-Adaptive Sparsity Pruning (LAMP) (<xref ref-type="bibr" rid="B21">Lee et&#xa0;al., 2020</xref>), which proposes a global pruning importance score. Global pruning is characterized by removing connections below the LAMP scores in the whole model rather than removing connections below a threshold score in each layer, i.e., global pruning is not equally sparse for each layer. As shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>, the LAMP scores are the square of the weight size, normalized by the total of all remaining weights in the layer.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Illustration of the LAMP (Layer-Adaptive Sparsity Pruning) scores.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g007.tif"/>
</fig>
<p>Consider a feedforward neural network of depth-d with corresponding weight tensors <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for each convolutional layer and fully connected layer. Each weight tensor is assumed to be expanded into a one-dimensional vector to define the LAMP scores uniform for both the fully connected and convolutional layers. For these one-dimensional vectors, we assume the weights are sorted depending on the index map in ascending order, i.e., <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> holds whenever <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the entries of <inline-formula>
<mml:math display="inline" id="im25">
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula> that are mapped by the index <inline-formula>
<mml:math display="inline" id="im26">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula>. The LAMP scores corresponding to the <inline-formula>
<mml:math display="inline" id="im27">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula>-th position in the weight tensor <inline-formula>
<mml:math display="inline" id="im28">
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula> are established in the <xref ref-type="disp-formula" rid="eq4">Equation (4)</xref>.</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:math>
</disp-formula>
<p>Once the LAMP scores are computed, the minimum-scoring connections are globally pruned until the required global sparsity constraints are satisfied. That is, for any given weight tensor <inline-formula>
<mml:math display="inline" id="im29">
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula>, along with indicators <inline-formula>
<mml:math display="inline" id="im30">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im31">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula>, the following conditions need to be satisfied in the <xref ref-type="disp-formula" rid="eq5">Equation (5)</xref>.</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&gt;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x21d2;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&gt;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>All connections with the LAMP scores less than the target weights are pruned. Global pruning using the LAMP scores is similar to MP-based hierarchical pruning with automatic selection of hierarchical sparsity. Pruning the model with the LAMP scores after training maintains the benefits of MP, and the LAMP scores do not depend on any model-specific knowledge, eliminating the need for a long, sparse training step. The overall process of pruning is depicted in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Model pruning flowchart.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g008.tif"/>
</fig>
</sec>
<sec id="s2_9">
<label>2.9</label>
<title>Block-Correlation Knowledge Distillation</title>
<p>Knowledge distillation is a classical approach to model compression, a concept initially put forward by Hinton et&#xa0;al (<xref ref-type="bibr" rid="B17">Hinton et&#xa0;al., 2015</xref>), and during the following years, researchers have proposed many more knowledge distillation methods, such as Logits distillation and Features distillation (<xref ref-type="bibr" rid="B1">Adriana et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B2">Ahn et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B16">Heo et&#xa0;al., 2019</xref>). The core idea of knowledge distillation is to extract knowledge from the better-performing and more complex structure of the teacher model and transfer the knowledge to the more lightweight student model without changing the network structure so that the performance and versatility of the smaller model can be enhanced.</p>
<p>Previous approaches obtained good results but performed poorly on small datasets. Therefore, we adopted BCKD (<xref ref-type="bibr" rid="B40">Wang et&#xa0;al., 2023</xref>). Unlike conventional Logits distillation or Features distillation, BCKD notices the connections between blocks in a neural network, providing new knowledge for distillation. This approach improves performance, does not introduce additional computational overhead, and addresses the problem of poor distillation on small datasets.</p>
<p>
<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> illustrates the structure of the BCKD. The classification task&#x2019;s bottom right corner is the cross-entropy loss function (CE). The bottom uses conventional knowledge distillation (KD) for the trained student model. Finally, the block correlation loss function (BC) is employed to augment the distillation&#x2019;s efficacy.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>BCKD (Block-Correlation Knowledge Distillation) structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g009.tif"/>
</fig>
<p>BCKD uses ResNet32x4 and ResNet8x4 (<xref ref-type="bibr" rid="B15">He et&#xa0;al., 2016</xref>) as infrastructure. Setting the set <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:mtext>B&#xa0;</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be the output candidates for each residual block from teachers and students, and the correlation between neighboring blocks is assumed to be a relationship that the model can learn, denoted as <inline-formula>
<mml:math display="inline" id="im33">
<mml:mrow>
<mml:mtext>C&#xa0;</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. All candidate outputs have their own feature mapping size and channel dimension, e.g., <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>&#x3001; <inline-formula>
<mml:math display="inline" id="im36">
<mml:mi>D</mml:mi>
</mml:math>
</inline-formula>&#x3001; <inline-formula>
<mml:math display="inline" id="im37">
<mml:mi>H</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im38">
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula> denote the batch size, channel dimension, height, and width, respectively. When the number of neighboring blocks is <inline-formula>
<mml:math display="inline" id="im39">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula>, it is clear that the number of correlations is <inline-formula>
<mml:math display="inline" id="im40">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. For ease of exposition, we use <inline-formula>
<mml:math display="inline" id="im41">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to denote the blocks and correlations of students and teachers, respectively. From this, we derive the equation for the correlation set <inline-formula>
<mml:math display="inline" id="im43">
<mml:mtext>C</mml:mtext>
</mml:math>
</inline-formula>:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>&#x3c8;</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>j</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
<mml:mi>&#xa0;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>&#x2299;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mtext>T</mml:mtext>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>j</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
<mml:mi>&#xa0;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>&#x2299;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mtext>T</mml:mtext>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="eq6">Equation (6)</xref> is a convolutional kernel self-adaptive function that is used to average the values of the spatial dimensions of the feature maps, whereby <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be unified according to the size of the feature maps of <inline-formula>
<mml:math display="inline" id="im46">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c8;</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> represents the average pooling of the channels, so the feature map sizes of <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> have feature mappings of equal size. The <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="eq7">Equation (7)</xref> represents the normalized softmax function, thus <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>j</mml:mtext>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>j</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Next, we utilize the multilayer perceptron (MLP) to help match <inline-formula>
<mml:math display="inline" id="im52">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im53">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with <inline-formula>
<mml:math display="inline" id="im54">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im55">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in order to preserve the features while trying to apply different network structures. Considering the important influence of the discriminative classifier on the model&#x2019;s detection capability, we input the results obtained from the MLPs into the classifiers of the teacher&#x2019;s model with the aim of obtaining a more comprehensive representation of the correlation features. The resultant <inline-formula>
<mml:math display="inline" id="im56">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im57">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are expressed are expressed in the <xref ref-type="disp-formula" rid="eq8">Equations (8)</xref>, <xref ref-type="disp-formula" rid="eq9">(9)</xref>.</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im58">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the linear layer, <inline-formula>
<mml:math display="inline" id="im59">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the ReLu activation function, <inline-formula>
<mml:math display="inline" id="im60">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> stands for L2 for normalization, <inline-formula>
<mml:math display="inline" id="im61">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the classifier of the teacher model, <inline-formula>
<mml:math display="inline" id="im62">
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> denotes <inline-formula>
<mml:math display="inline" id="im63">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>j</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula>
<mml:math display="inline" id="im64">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>j</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>. In <inline-formula>
<mml:math display="inline" id="im65">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im66">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im67">
<mml:mi>C</mml:mi>
</mml:math>
</inline-formula> is the number of elements in the set <inline-formula>
<mml:math display="inline" id="im68">
<mml:mtext>C</mml:mtext>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im69">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the batch size, and <inline-formula>
<mml:math display="inline" id="im70">
<mml:mi>M</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im71">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula> stand for the output dimensions of <inline-formula>
<mml:math display="inline" id="im72">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im73">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively.</p>
<p>After the above process, the correlations of the student model and the teacher model are then mapped into the same feature space. In this feature space, the correlation between neighboring blocks of the pre-trained teacher model is strong, and the distribution of different samples is consistent. In contrast, the untrained student model is divergent across samples. Therefore, for the student model to understand the gap between itself and the teacher <inline-formula>
<mml:math display="inline" id="im74">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im75">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>L</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, it can learn more knowledge and thus improve the performance of the student itself.</p>
<p>Regarding loss function, if L1 (MAE) or L2 (MSE) is used directly, it does not work well for the untrained student model. The MSE loss function performs better for the model in terms of gradient and convergence, and the MAE loss function performs more consistently when dealing with outliers. Therefore, BCKD chose Huber loss (<xref ref-type="bibr" rid="B13">Gokcesu and Gokcesu, 2021</xref>) to combine the respective advantages of MAE and MSE in order to get better performance from the student model.</p>
</sec>
<sec id="s2_10">
<label>2.10</label>
<title>Pseudo code for combinatorial algorithms</title>
<p>By integrating the approaches in the above subsections, we designed an algorithm for Color-changing melon maturity detection, divided into five main steps: initialization, model training, model pruning, knowledge distillation, and testing. The algorithm enhances the robustness of the model when dealing with complex scenarios while maintaining the lightweight characteristics of the model, and the pseudo-code is given in <xref ref-type="statement" rid="algo1">
<bold>Algorithm 1</bold>
</xref>.</p>
<statement id="algo1">
<label>Algorithm 1</label>
<title>Pseudo code for combinatorial algorithms.</title>
<p>
<preformat>
<bold>Input:</bold> Color-changing melon images, pretrained YOLOv8s weight
<bold>Output:</bold> Category confidence and prediction frame coordinates
1: <bold>Initialize:</bold> img_split=img_train(80%) + img_val(80%) + img_test(80%);
2: <bold>Training on Server:</bold>
3: batch_size=64, img_size=640, epochs E<sub>t</sub>= 300;
4: <bold>for</bold> <italic>i</italic> = 1: <italic>E<sub>t</sub>
</italic> <bold>do</bold>
5:  Train on the img_train with the improved YOLOv8;
6:  Calculate the loss function;
7:  Evaluate model using img_val;
8: <bold>end for</bold>
9: save best_weight.pt;
10: <bold>Model Prune on Server</bold>
11: model=best_weight.pt, epochs E<sub>p</sub>= 9999, pruned_method=LAMP, speed_up=3.0;
12: <bold>for</bold> <italic>i</italic> = 1: <italic>E<sub>p</sub>
</italic> <bold>do</bold>
13:  if(speed_up &gt; 3.0)
14:  break;
15:  Calculate the LAMP score for each channel in the improved model;
16:  Remove the connection with the minimum score and calculate speed_up;
17: <bold>end for</bold>
18: save last_prune.pt;
19: <bold>Knowledge Distillation on Server</bold>
20: model= last_prune.pt, epochs E<sub>kd</sub>= 250, kd_method=BCKD, teacher=YOLOv8s-Improved;
21: <bold>for</bold> <italic>i</italic> = 1: <italic>E<sub>kd</sub>
</italic> <bold>do</bold>
22:  Predict training data and generate sample distribution space;;
23:  Utilize BCKD to help students calculate the difference between their own and their teacher&#x2019;s predictions of outcomes;
24:  Update the parameters of the student model using the loss function;
25: <bold>end for</bold>
26: save best_kd_weight.pt;
27: <bold>Testing on laptop</bold>
28: Predict model using img_test;
29: Obtain the output result.
</preformat>
</p>
</statement>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and discussion</title>
<sec id="s3_1">
<label>3.1</label>
<title>Experimental setup</title>
<p>This study&#x2019;s experiments were all performed on an Ubuntu 18.04 system; the programming language was Python 3.9.16, and the network framework used was Pytorch 1.10.0 (cuda 11.7). For our training phase, we used a high-performance server configured with an Intel<sup>&#xae;</sup> i7 13700K 16C5.40GHz CPU and an NVIDIA RTX 4090 GPU. For the inference testing phase, we used an Intel<sup>&#xae;</sup> i7 8750H 4C2.20GHz and an NVIDIA GTX 1050ti laptop to simulate a resource-constrained device and test the model&#x2019;s performance. The hyperparameter settings for the training phase include a batch size of 64 and 300 epochs for the number of training rounds, and the officially provided pre-training weights are used as the initial weights. The rest of the hyperparameters are the default values of YOLOv8.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Evaluation indicators</title>
<p>In this study, a total of seven metrics, namely, precision, recall, average precision, model size, number of Params, FLOPs, and FPS, are used to comprehensively evaluate the performance of the model.</p>
<p>The formulas for precision and recall are expressed in the <xref ref-type="disp-formula" rid="eq10">Equations (10)</xref>, <xref ref-type="disp-formula" rid="eq11">(11)</xref>.</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the above two equations, <inline-formula>
<mml:math display="inline" id="im76">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> stands for the number of targets correctly judged as positive, <inline-formula>
<mml:math display="inline" id="im77">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> stands for the number of targets incorrectly judged as positive, and <inline-formula>
<mml:math display="inline" id="im78">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> stands for the number of targets belonging to positive but incorrectly judged as negative. <inline-formula>
<mml:math display="inline" id="im79">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is used to calculate the average precision of a single class of targets under different recall rates, and <inline-formula>
<mml:math display="inline" id="im80">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is used to calculate the average <inline-formula>
<mml:math display="inline" id="im81">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of multiple classes of targets, and their definitions are expressed in the <xref ref-type="disp-formula" rid="eq12">Equations (12)</xref>, <xref ref-type="disp-formula" rid="eq13">(13)</xref>, respectively:</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mstyle>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>For all kinds of targets detected, the higher the <inline-formula>
<mml:math display="inline" id="im82">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the more accurate the model&#x2019;s predictions are; therefore, it represents a better detection performance of the model. <inline-formula>
<mml:math display="inline" id="im83">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>@</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> denotes the average <inline-formula>
<mml:math display="inline" id="im84">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of each kind of target when the IoU threshold is set to 0.5. <inline-formula>
<mml:math display="inline" id="im85">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>@</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>:</mml:mo>
<mml:mn>0.95</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> The thresholds are computed from the range of 0.5 to 0.95, with an increase of 0.05 in each step, and the obtained average <inline-formula>
<mml:math display="inline" id="im86">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of each kind of the average <inline-formula>
<mml:math display="inline" id="im87">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the targets, they are defined in the <xref ref-type="disp-formula" rid="eq14">Equations (14)</xref>, <xref ref-type="disp-formula" rid="eq15">(15)</xref>, respectively:</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>@</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>@</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>:</mml:mo>
<mml:mn>0.95</mml:mn>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>j</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.55</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.6</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>0.95</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Params denote the sum of parameters to be trained within the model. FLOPs denote the number of floating point operations required during network training, and model size denotes the size of the model. The lower these three metrics are, the more lightweight the model is and, therefore, the more suitable it is for deployment on edge devices.</p>
<p>Frames per second (<inline-formula>
<mml:math display="inline" id="im88">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) measure the model&#x2019;s detection speed. <inline-formula>
<mml:math display="inline" id="im89">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is calculated from the inverse sum of the pre-processing time, inference time, and post-processing time. The larger its value, the faster the real-time detection of the model. Its definition is in the <xref ref-type="disp-formula" rid="eq16">Equation (16)</xref>.</p>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mtext>inf</mml:mtext>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Comparison of before and after improvements</title>
<p>We trained the base YOLOv8s and the improved YOLOv8 separately with the same parameter settings on the server side and then compared the hotspots of attention of the two models under four scenarios on the test set on the laptop side. As shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, when the fruits are denser, the improved model shows a higher degree of hotness for the fruit-concentrated regions, while the hotspots of the base model are more scattered. When the fruits are more dispersed, the improved model has more hotspots and a little more heat for the regions where the fruits are located. For regions where the fruits overlap, the improved model pays more attention to the occluded fruits, while the base model pays less attention to the occluded fruits. The final image shows the similarity between the background and the fruit, which is similar to the common mimicry in the insect world, and it can be seen that the base model mistook the white pipe on the right side for an immature fruit while the improved model shows the hotspots of attention to the fruit very well.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Comparison of visualized heat maps in different scenarios.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g010.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> shows some of the detection results of the two models on the test set. </p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Comparison of detection effect.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g011.tif"/>
</fig>
<p>In the first row of Color-changing melon detection, we list the detection effects of the two models when the usual scene, backlight, and fruit overlap, respectively. Both models perform better, but the confidence of the improved model is generally higher than that of YOLOv8s, and there are no misdetected fruits. When facing smaller fruits and fruits at the edge of the figure, the improved model has better detection performance, and the confidence level is 0.28 higher than that of YOLOv8s. When switching to the second row of Color-changing melon detection, we compared the detection effects of the two models when the fruits are at the edge of the figure, dense fruits, and fruits are obscured by other objects, and the confidence level of the improved model is still higher than that of YOLOv8s, especially when other objects obscure the fruits. They are more effective when other objects occlude them. However, given the rigor of the article, we also give examples of rare errors in the detection process of both models, as the co-obscuration of leaves and other objects creates a visual misalignment, which results in a situation where both models detect the same fruit as two targets.</p>
<p>Overall, YOLOv8s misdetected the background as fruit in a few cases and had problems with ambiguous judgments about the maturity of some fruits. In contrast, the improved model is more accurate in localizing and classifying fruits with a higher confidence level.</p>
<p>In order to illustrate more intuitively the prediction accuracy of the three categories of fruit maturity before and after the model improvement, we present the normalized confusion matrix of the training results. As depicted in <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>, in the confusion matrix, the rows indicate the predicted labels for the categories, and the columns indicate the true labels for the categories. In each square, darker colors indicate larger values, lighter colors indicate lower values, and white indicates empty values. By looking at the differences between the predicted values for each category, it can be observed that the values of the improved model&#x2019;s diagonal lines add up to a larger sum and improve the accuracy of the predictions for semi-mature fruits, as well as reduce the proportion of backgrounds that are mistakenly detected as immature fruits. This demonstrates that the multilevel feature fusion mechanism we constructed reduces the interference of complex background on fruit detection accuracy to some extent.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Comparison of confusion matrices.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g012.tif"/>
</fig>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Results of ablation experiments</title>
<p>The server&#x2019;s ablation experimental results are recorded in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The first row uses YOLOv8s as the baseline, and each module can be added to the model independently. Where A denotes the use of the DWR-DRB module, B stands for the use of the HS-PAN module, and A+B denotes the merging of the two model optimization methods into the base model. As seen from the table, each module suggested in this research contributes to the model performance when merged into the base model and reduces the network structure&#x2019;s complexity and the associated computational workload.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Results of ablation experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">A</th>
<th valign="top" align="center">B</th>
<th valign="top" align="center">A + B</th>
<th valign="middle" align="center">mAP@0.5</th>
<th valign="top" align="center">mAP@0.5:0.95</th>
<th valign="top" align="center">Precision</th>
<th valign="top" align="center">Recall</th>
<th valign="top" align="center">Params</th>
<th valign="top" align="center">GFLOPs</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="middle" align="center">94.8%</td>
<td valign="top" align="center">86.5%</td>
<td valign="middle" align="center">98.6%</td>
<td valign="middle" align="center">97%</td>
<td valign="top" align="center">11.13 M</td>
<td valign="top" align="center">28.4</td>
</tr>
<tr>
<td valign="top" align="center">&#x221a;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="middle" align="center">95.0%</td>
<td valign="top" align="center">88.6%</td>
<td valign="top" align="center">97.7%</td>
<td valign="top" align="center">98%</td>
<td valign="top" align="center">10.46 M</td>
<td valign="top" align="center">27.4</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x221a;</td>
<td valign="top" align="center"/>
<td valign="middle" align="center">96.2%</td>
<td valign="top" align="center">87.1%</td>
<td valign="top" align="center">96.6%</td>
<td valign="top" align="center">99%</td>
<td valign="top" align="center">7.73 M</td>
<td valign="top" align="center">25.0</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x221a;</td>
<td valign="middle" align="center">95.5%</td>
<td valign="top" align="center">89.0%</td>
<td valign="top" align="center">98.3%</td>
<td valign="top" align="center">99%</td>
<td valign="top" align="center">6.47 M</td>
<td valign="top" align="center">22.8</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The "&#x221a;" symbol indicates that the module is used.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Results after model pruning</title>
<p>After several experiments, we adopted a pruning acceleration ratio of 3.0 for the improved model because the pruning rate at this point can make the mAP stay relatively good. During the neural network&#x2019;s pruning phase, the layer with the high LAMP scores has a higher importance level of its channels. Thus, a small amount of pruning or directly skipping the pruning of the channels of that layer can be done to remove the non-essential connections more reasonably to maintain the performance of the pruned model. <xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref> demonstrates the changes in the number of channels in each layer of the pruned model. It can be seen that some of the convolutional layers are pruned strongly, while the float in the number of channels in most of the layers is not significant. After pruning, the range of channel counts changed from 1 to 512 before pruning to 1 to 74 after pruning. mAP 0.5 and mAP@0.5:0.95 were reduced by 0.8% and 2.9%, respectively. The number of model Params decreased from 6.47 M to 1.14 M, a reduction of 82%. The number of computed FLOPs is reduced from 22.8 GFLOPs to 7.5 GFLOPs, a 67% reduction. The model size is also changed from 12.64 MB to 2.47 MB.</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>Comparison before and after channel compression.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g013.tif"/>
</fig>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Effect of different teacher models on distillation</title>
<p>After a model is pruned, the accuracy generally decreases by a certain degree, and it is generally essential to adjust the pruned model accordingly with the aim of restoring its accuracy. Since the structure of the pruned model is relatively simple, it has good portability while ensuring prediction accuracy when the task goal is relatively clear. We distill the pruned model as a student model using the BCKD method. To evaluate the impact of various size scales of teachers on the effect of the student models, we trained the students on their knowledge using the base YOLOv8 and the improved model in s, m, and l sizes, respectively. <xref ref-type="fig" rid="f14">
<bold>Figure&#xa0;14</bold>
</xref> illustrates the validation results on a laptop after distillation training with different teacher models. <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> provides specific data after validation of the distillation models. I-YOLOv8 in the table represents the improved YOLOv8 model.</p>
<fig id="f14" position="float">
<label>Figure&#xa0;14</label>
<caption>
<p>Validation results of the teacher model at different scales.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1406593-g014.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Validation results of distillation by different teachers.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Teacher-model</th>
<th valign="middle" align="center">mAP@0.5</th>
<th valign="top" align="center">mAP@0.5:0.95</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="top" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLOv8s</td>
<td valign="middle" align="center">95.5%</td>
<td valign="middle" align="center">86.1%</td>
<td valign="middle" align="center">56.2%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">63.8</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8m</td>
<td valign="middle" align="center">95.1%</td>
<td valign="top" align="center">85.9%</td>
<td valign="middle" align="center">57.4%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">63.8</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8l</td>
<td valign="middle" align="center">94.0%</td>
<td valign="top" align="center">84.6%</td>
<td valign="middle" align="center">58.9%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">63.6</td>
</tr>
<tr>
<td valign="middle" align="center">I-YOLOv8s</td>
<td valign="middle" align="center">96.1%</td>
<td valign="top" align="center">88.1%</td>
<td valign="middle" align="center">97.8%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">78.9</td>
</tr>
<tr>
<td valign="middle" align="center">I-YOLOv8m</td>
<td valign="middle" align="center">95.6%</td>
<td valign="top" align="center">86.6%</td>
<td valign="middle" align="center">95.3%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">78.7</td>
</tr>
<tr>
<td valign="middle" align="center">I-YOLOv8l</td>
<td valign="middle" align="center">95.3%</td>
<td valign="top" align="center">86.3%</td>
<td valign="middle" align="center">95.9%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">78.4</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>According to the comparison results in the table, it is easy to see that the base model of different sizes is not as effective in training the students as the improved model. Furthermore, the distillation of the student model by the untrained base model is too low in accuracy, and there is a cliff drop in detection speed. While the teacher models themselves continue to improve, they perform less and less well for distillation training. Compared to the other models as teachers, Improved-YOLOv8s had the most outstanding results for student training when the student model achieved 96.1% mAP 0.5 and 88.1% mAP@0.5:0.95. YOLOv8l had the worst distillation training as a teacher when the student model achieved 94.1% mAP0.5 and 88.1% mAP 0.5 only 94.1% and mAP@0.5:0.95 only 84.6%. This indicates that when the gap between the teacher and student models is within a reasonable range, the student model obtains more effective knowledge from the teacher network. On the contrary, when the gap between the two is too large, the student model is unable to learn much of the knowledge from the teacher&#x2019;s network, and the training effect is not as good as expected. Therefore, we choose the s version of the improved model as the teacher network to distill the pruned model.</p>
</sec>
<sec id="s3_7">
<label>3.7</label>
<title>Comparison between different target detection networks</title>
<p>To evaluate the efficacy of our suggested approach, we trained various lightweight networks in the same server-side experimental environment. We tested the trained models on a laptop to simulate the environment of resource-constrained hardware. The specific experimental results are shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. In the table, I-YOLOv8 represents the improved YOLOv8 model, I-P-YOLOv8 represents the improved model trained with fine-tuning after pruning, and I-P-KD-YOLOv8 represents the improved model trained with knowledge distillation after pruning. Under the condition of mAP@0.5, I-YOLOv8 and I-P-KD-YOLOv8 showed better results compared to other models. When the evaluation metric becomes the more stringent mAP@0.5:0.95, the models with a larger number of Params and computed FLOPs take advantage, leading the other lightweight networks by 5% to 10%, which can be seen in the good performance of YOLOv8s as well as our subsequent improved models, and among them I-YOLOv8 is also 2.5% higher than YOLOv8s. In terms of precision and recall, the performance of the various lightweight networks does not differ much. After pruning, the FLOPs and Params of I-P-YOLOv8 are significantly lower than other lightweight networks and even smaller than YOLOv8n. Meanwhile, the detection speed of I-P-YOLOv8 is somewhat improved, which is 9.1% and 14% faster than YOLOv8s and I-YOLOv8, respectively. Subsequently, after knowledge distillation, I-P-KD-YOLOv8 shows a large improvement in the metrics of mAP, and the overall performance outperforms that of I-P-YOLOv8. It can be seen that I-P-KD-YOLOv8 maintains the detection performance of I-YOLOv8. Meanwhile, the number of parameters and FLOPs are significantly reduced, and the model size is the smallest, which is an excellent balance between accuracy, speed, and model lightweight. YOLOv8s has higher detection accuracy than YOLOv8n, while the parameters and computational effort are much smaller than YOLOv8m. S-scale models can strike a good balance between detection performance and model complexity relative to n- and m-scale YOLOv8 models, so we chose YOLOv8s as the original model and improved it. After subsequent model pruning and knowledge distillation operations, the detection accuracy of the improved YOLOv8s is higher than that of YOLOv8n. However, it is more lightweight, making it more suitable for robot picking for automated Color-changing melons.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison of different target detection networks.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="center">mAP@0.5</th>
<th valign="top" align="center">mAP@0.5:0.95</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="top" align="center">Params</th>
<th valign="top" align="center">GFLOPs</th>
<th valign="top" align="center">FPS</th>
<th valign="top" align="center">Size</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">YOLOv4-tiny</td>
<td valign="middle" align="center">94.4%</td>
<td valign="middle" align="center">74.5%</td>
<td valign="middle" align="center">93.8%</td>
<td valign="middle" align="center">93%</td>
<td valign="middle" align="center">5.88 M</td>
<td valign="middle" align="center">16.2</td>
<td valign="top" align="center">52.0</td>
<td valign="middle" align="center">22.4 MB</td>
</tr>
<tr>
<td valign="middle" align="left">YOLOv5s</td>
<td valign="middle" align="center">93.5%</td>
<td valign="top" align="center">78.6%</td>
<td valign="middle" align="center">93.8%</td>
<td valign="middle" align="center">96%</td>
<td valign="top" align="center">7.03 M</td>
<td valign="top" align="center">16.0</td>
<td valign="top" align="center">55.1</td>
<td valign="top" align="center">14.5 MB</td>
</tr>
<tr>
<td valign="middle" align="left">YOLOv7-tiny</td>
<td valign="middle" align="center">93.4%</td>
<td valign="middle" align="center">70.8%</td>
<td valign="middle" align="center">94.0%</td>
<td valign="middle" align="center">99%</td>
<td valign="middle" align="center">6.01 M</td>
<td valign="middle" align="center">13.1</td>
<td valign="top" align="center">66.5</td>
<td valign="middle" align="center">12.3 MB</td>
</tr>
<tr>
<td valign="middle" align="left">YOLOv8n</td>
<td valign="middle" align="center">94.1%</td>
<td valign="top" align="center">81.2%</td>
<td valign="middle" align="center">95.7%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">3.01 M</td>
<td valign="middle" align="center">8.1</td>
<td valign="top" align="center">64.8</td>
<td valign="top" align="center">6.2 MB</td>
</tr>
<tr>
<td valign="middle" align="left">YOLOv8s</td>
<td valign="middle" align="center">94.8%</td>
<td valign="top" align="center">86.5%</td>
<td valign="middle" align="center">98.6%</td>
<td valign="middle" align="center">97%</td>
<td valign="top" align="center">11.13 M</td>
<td valign="middle" align="center">28.4</td>
<td valign="top" align="center">72.3</td>
<td valign="top" align="center">21.5 MB</td>
</tr>
<tr>
<td valign="middle" align="left">I-YOLOv8</td>
<td valign="middle" align="center">95.5%</td>
<td valign="top" align="center">89.0%</td>
<td valign="middle" align="center">98.3%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">6.47 M</td>
<td valign="middle" align="center">22.8</td>
<td valign="top" align="center">69.2</td>
<td valign="top" align="center">12.6 MB</td>
</tr>
<tr>
<td valign="middle" align="left">I-P-YOLOv8</td>
<td valign="middle" align="center">94.7%</td>
<td valign="top" align="center">86.1%</td>
<td valign="middle" align="center">96.0%</td>
<td valign="middle" align="center">99%</td>
<td valign="top" align="center">1.14M</td>
<td valign="middle" align="center">7.5</td>
<td valign="top" align="center">78.9</td>
<td valign="top" align="center">2.5 MB</td>
</tr>
<tr>
<td valign="middle" align="left">I-P-KD-YOLOv8</td>
<td valign="top" align="center">96.1%</td>
<td valign="top" align="center">88.1%</td>
<td valign="top" align="center">97.8%</td>
<td valign="top" align="center">99%</td>
<td valign="top" align="center">1.14M</td>
<td valign="top" align="center">7.5</td>
<td valign="top" align="center">78.9</td>
<td valign="top" align="center">2.5 MB</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusion</title>
<p>In this study, we focus on the need for robotic picking of Color-changing melons and put the front-loaded fruit detection work into the study, which lays the foundation for further realization of automatic picking work. We first designed the DWR module and DRB to expand the receptive field, aiming to further strengthen the Backbone part&#x2019;s ability to acquire multi-scale contextual information. Subsequently, we design HS-PAN with multi-level feature fusion to strengthen the localization information and enrich the semantic information in the feature fusion process, which helps to enhance the detection network&#x2019;s attention to Color-changing melons&#x2019; details, thus increasing the accuracy of fruit detection and reducing the proportion of false detections. Then, we simplified the improved network structure by pruning unimportant connections in the detection network using Layer-Adaptive Sparsity Pruning. Finally, the accuracy of the pruning model is further recovered using Block-Correlation Knowledge Distillation and compared to other lightweight networks. To summarize, we first improve and train modules on larger models for complex scenarios. Next, we drastically simplify the network structure of the improved model by model pruning. Finally, we make the accuracy of the pruned model close to the pre-pruning level by knowledge distillation. The advantages of our proposed combinatorial algorithm are that it obtains higher accuracy and stronger robustness than other lightweight models. In contrast, the complexity of the final model is much lower than that of the lightweight networks. The generalization of our proposed approach is that it is more conducive to achieving deployment on edge devices by utilizing small models with superior performance. Although the effectiveness of our proposed combined algorithm for Color-changing melon maturity detection has been validated, some things could be improved. The current algorithm is specific to Color-changing melon fruits, and the detection performance for more fruit varieties still needs to be proven. Meanwhile, this study only deals with the mature recognition part of Color-changing melon and does not mention the algorithms related to localization in the robotic picking behavior. In conclusion, our proposed algorithm can provide technical support for picking color-changing melons and some ideas for the automatic picking of melons, which need to be researched more in intelligent agriculture. In the future, we can explore other application scenarios, such as the detection of tomato fruits, to verify the applicability of the algorithm in other scenarios.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>GC: Formal analysis, Methodology, Supervision, Writing &#x2013; original draft. YH: Conceptualization, Data curation, Methodology, Resources, Software, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. HC: Validation, Visualization, Writing &#x2013; review &amp; editing. LC: Supervision, Validation, Writing &#x2013; review &amp; editing. JY: Investigation, Methodology, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adriana</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Nicolas</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ebrahimi</surname> <given-names>K. S.</given-names>
</name>
<name>
<surname>Antoine</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Carlo</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Yoshua</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Fitnets: Hints for thin deep nets</article-title>. <source>Mach. Learn.</source> <volume>2</volume>, <elocation-id>1</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1412.6550</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ahn</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>S. X.</given-names>
</name>
<name>
<surname>Damianou</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Lawrence</surname> <given-names>N. D.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Variational information distillation for knowledge transfer</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>9163</fpage>&#x2013;<lpage>9171</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1904.05835</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Yolov4: Optimal speed and accuracy of object detection</article-title>. <source>Comput. Vision Pattern Recognition</source>. <volume>21</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2004.10934</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Camposeo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Vivaldi</surname> <given-names>G. A.</given-names>
</name>
<name>
<surname>Gattullo</surname> <given-names>C. E.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Ripening indices and harvesting times of different olive cultivars for continuous harvest</article-title>. <source>Scientia Hortic.</source> <volume>151</volume>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scienta.2012.12.019</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chaudhari</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Waghmare</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Machine vision based fruit classification and grading&#x2014;a review</article-title>,&#x201d; in <source>
<italic>ICCCE 2021</italic>: Proceedings of the 4th International Conference on Communications and Cyber Physical Engineering</source> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>775</fpage>&#x2013;<lpage>781</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-981-16-7985-8_81</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>C.-F.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Mallinar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Sercu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Feris</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Big-little net: An efficient multi-scale feature representation for visual and speech recognition</article-title>. <source>Comput. Vision Pattern Recognition</source>. <volume>1807.03848</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1807.03848</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Accurate leukocyte detection based on deformable-DETR and multi-level feature fusion for aiding diagnosis of blood diseases</article-title>. <source>Comput. Biol. Med.</source> <volume>170</volume>, <elocation-id>107917</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.107917</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clark</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>G. D.</given-names>
</name>
<name>
<surname>Kendall</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Taylor</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>A proposed framework for accelerating technology trajectories in agriculture: A case study in China</article-title>. <source>Front. Agric. Sci. Eng.</source> <volume>5</volume>, <fpage>485</fpage>&#x2013;<lpage>498</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.15302/J-FASE-2018244</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Unireplknet: A universal perception large-kernel convnet for audio, video, point cloud, time-series and image recognition</article-title>. <source>Comput. Vision Pattern Recognition</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2311.15599</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dorj</surname> <given-names>U.-O.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yun</surname> <given-names>S.-s.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>An yield estimation in citrus orchards via fruit detection and counting using image processing</article-title>. <source>Comput. Electron. Agric.</source> <volume>140</volume>, <fpage>103</fpage>&#x2013;<lpage>112</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2017.05.019</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gale</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Elsen</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Hooker</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The state of sparsity in deep neural networks</article-title>. <source>Mach. Learn</source>. <volume>1902.09574</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1902.09574</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gale</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zaharia</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Young</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Elsen</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Sparse gpu kernels for deep learning</article-title>,&#x201d; in <conf-name>SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, IEEE</conf-name>. <fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/SC41405.2020.00021</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gokcesu</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Gokcesu</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Generalized huber loss for robust learning and its efficient minimization for a robust statistics</article-title>. <source>Mach. Learn</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2108.12627</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Gkioxari</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Mask r-cnn</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE international conference Computer Vision and Pattern Recognition</conf-name>. <fpage>2961</fpage>&#x2013;<lpage>2969</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1703.06870</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>770</fpage>&#x2013;<lpage>778</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1512.03385</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Heo</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yun</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>J. Y.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Knowledge transfer via distillation of activation boundaries formed by hidden neurons</article-title>,&#x201d; in <conf-name>Proceedings of the AAAI conference on artificial intelligence</conf-name>. <fpage>3779</fpage>&#x2013;<lpage>3787</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1609/aaai.v33i01.33013779</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hinton</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Vinyals</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Dean</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Distilling knowledge Neural network</article-title>. <source>Mach. Learn</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1503.02531</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamilaris</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Prenafeta-Bold&#xfa;</surname> <given-names>F. X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep Learn. agriculture: A survey</article-title>. <source>Comput. Electron. Agric</source> <volume>147</volume>, <fpage>70</fpage>&#x2013;<lpage>90</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2018.02.016</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Real-time fruit recognition and grasping estimation for robotic apple harvesting</article-title>. <source>Sensors</source> <volume>20</volume>, <elocation-id>5670</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s20195670</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krizhevsky</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>ImageNet classification with deep convolutional neural networks</article-title>. <source>Commun. ACM</source> <volume>60</volume>, <fpage>84</fpage>&#x2013;<lpage>90</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3065386</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Mo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ahn</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Layer-adaptive sparsity for the magnitude-based pruning</article-title>. <source>Mach. Learn</source>. <volume>2010.07611</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2010.07611</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lei</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Survey of deep neural network model compression</article-title>. <source>J. software</source> <volume>29</volume>, <fpage>251</fpage>&#x2013;<lpage>266</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.13328/j.cnki.jos.005428</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L. J. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Design and experiment of a broken corn kernel detection device based on the yolov4-tiny algorithm</article-title>. <source>Agriculture</source> <volume>11</volume>, <elocation-id>1238</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture11121238</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Generalized focal loss: Learning qualified and distributed bounding boxes for dense object detection</article-title>. <source>Comput. Vision Pattern Recognition</source> <volume>33</volume>, <fpage>21002</fpage>&#x2013;<lpage>21012</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2006.04388</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>A visual detection method for nighttime litchi fruits and fruiting stems</article-title>. <source>Comput. Electron. Agric.</source> <volume>169</volume>, <elocation-id>105192</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2019.105192</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>T.-Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Hariharan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Belongie</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Feature pyramid networks for object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>2117</fpage>&#x2013;<lpage>2125</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1612.03144</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Anguelov</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Erhan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Reed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>C.-Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). &#x201c;<article-title>Ssd: Single shot multibox detector</article-title>,&#x201d; in <conf-name>Computer Vision&#x2013;ECCV 2016: 14th European Conference</conf-name>, <conf-loc>Amsterdam, The Netherlands</conf-loc>, <conf-date>October 11&#x2013;14, 2016</conf-date>. <fpage>21</fpage>&#x2013;<lpage>37</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-46448-0_2</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Learning efficient convolutional networks through network slimming</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE international conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>2736</fpage>&#x2013;<lpage>2744</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1708.06519</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Path aggregation network for instance segmentation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>8759</fpage>&#x2013;<lpage>8768</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1803.01534</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>T.-S.</given-names>
</name>
<name>
<surname>Ninomiya</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>W. J. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Intact detection of highly occluded immature tomatoes on plants using deep learning techniques</article-title>. <source>Sensors</source> <volume>20</volume>, <elocation-id>2984</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s20102984</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>Y. J. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Intelligent detection of Multi-Class pitaya fruits in target picking row based on WGB-YOLO network</article-title>. <source>Comput. Electron. Agric.</source> <volume>208</volume>, <elocation-id>107780</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.107780</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nouaze</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Sikati</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>YOLO-appleScab: A deep learning approach for efficient and accurate apple scab detection in varied lighting conditions using CARAFE-enhanced YOLOv7</article-title>,&#x201d; in <conf-name>Biology and Life Sciences Forum, MDPI</conf-name>. <fpage>6</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/IOCAG2023-16688</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Divvala</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>You only look once: Unified, real-time object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>779</fpage>&#x2013;<lpage>788</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1506.02640</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Faster r-cnn: Towards real-time object detection with region proposal networks</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>28</volume>, <fpage>1137</fpage>&#x2013;<lpage>1149</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2016.2577031</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yamaguchi</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An attribution-based pruning method for real-time mango detection with YOLO network</article-title>. <source>Comput. Electron. Agric.</source> <volume>169</volume>, <elocation-id>105214</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105214</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Suddapalli</surname> <given-names>S. R.</given-names>
</name>
<name>
<surname>Shyam</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Using mask-RCNN to identify defective parts of fruits and vegetables</article-title>,&#x201d; in <source>Intelligent Human Computer Interaction</source> (<publisher-name>Springer</publisher-name>), <fpage>637</fpage>&#x2013;<lpage>646</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-98404-5_58</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Suo</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Dhupia</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Improved multi-classes kiwifruit detection in orchard to avoid collisions during robotic picking</article-title>. <source>Comput. Electron. Agric.</source> <volume>182</volume>, <elocation-id>106052</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2021.106052</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Terven</surname> <given-names>J.</given-names>
</name>
<name>
<surname>C&#xf3;rdova-Esparza</surname> <given-names>D.-M.</given-names>
</name>
<name>
<surname>Romero-Gonz&#xe1;lez</surname> <given-names>J.-A.J.M.L.</given-names>
</name>
<name>
<surname>Extraction</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A comprehensive review of yolo architectures in computer vision: From yolov1 to yolov8 and yolo-nas</article-title>. <source>Mach. Learn. Knowledge Extraction</source> <volume>5</volume>, <fpage>1680</fpage>&#x2013;<lpage>1716</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/make5040083</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <fpage>7464</fpage>&#x2013;<lpage>7475</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2207.02696</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>BCKD: block-correlation knowledge distillation</article-title>,&#x201d; in <conf-name>2023 IEEE International Conference on Image Processing (ICIP)</conf-name>. <fpage>3225</fpage>&#x2013;<lpage>3229</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICIP49359.2023.10222195</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>DWRSeg: rethinking efficient acquisition of multi-scale contextual information for real-time semantic segmentation</article-title>. <source>Comput. Vision Pattern Recognition</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2212.01173</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>X.</given-names>
</name>
<name>
<surname>An</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An blueberry ripeness detection model based on enhanced detail feature and content-aware reassembly</article-title>. <source>Agronomy</source> <volume>13</volume>, <elocation-id>1613</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agronomy13061613</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Lightweight tomato real-time detection method based on improved YOLO and mobile deployment</article-title>. <source>Comput. Electron. Agric.</source> <volume>205</volume>, <elocation-id>107625</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.107625</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>He</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>He</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Complete and accurate holly fruits counting using YOLOX object detection</article-title>. <source>Comput. Electron. Agric.</source> <volume>198</volume>, <elocation-id>107062</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.107062</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Enhancing geometric factors in model learning and inference for object detection and instance segmentation</article-title>. <source>IEEE Trans. Cybernetics</source> <volume>52</volume>, <fpage>8574</fpage>&#x2013;<lpage>8586</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TCYB.2021.3095305</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Y. J. C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Research on CBF-YOLO detection model for common soybean pests in complex environment</article-title>. <source>Comput. Electron. Agric.</source> <volume>216</volume>, <elocation-id>108515</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108515</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>