<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2024.1468188</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An improved YOLOv8n-IRP model for natural rubber tree tapping surface detection and tapping key point positioning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Xirui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Weiqiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2795716"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Junxiao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Ruiwu</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Xuanli</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Yongqi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Zhifu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1994634"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Mechanical and Electrical Engineering, Hainan University</institution>, <addr-line>Haikou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Information and Communication Engineering, Hainan University</institution>, <addr-line>Haikou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Wen-Hao Su, China Agricultural University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Yunchao Tang, Dongguan University of Technology, China</p>
<p>Sayantan Sarkar, Texas A&amp;M AgriLife Research, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Zhifu Zhang, <email xlink:href="mailto:996099@hainanu.edu.cn">996099@hainanu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>30</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1468188</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>07</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Zhang, Ma, Liu, Xu, Chen, Liu and Zhang</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhang, Ma, Liu, Xu, Chen, Liu and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Aiming at the problem that lightweight algorithm models are difficult to accurately detect and locate tapping surfaces and tapping key points in complex rubber forest environments, this paper proposes an improved YOLOv8n-IRP model based on the YOLOv8n-Pose. First, the receptive field attention mechanism is introduced into the backbone network to enhance the feature extraction ability of the tapping surface. Secondly, the AFPN structure is used to reduce the loss and degradation of the low-level and high-level feature information. Finally, this paper designs a dual-branch key point detection head to improve the screening ability of key point features in the tapping surface. In the detection performance comparison experiment, the YOLOv8n-IRP improves the D_mAP50 and P_mAP50 by 1.4% and 2.3%, respectively, over the original model while achieving an average detection success rate of 87% in the variable illumination test, which demonstrates enhanced robustness. In the positioning performance comparison experiment, the YOLOv8n-IRP achieves an overall better localization performance than YOLOv8n-Pose and YOLOv5n-Pose, realizing an average Euclidean distance error of less than 40 pixels. In summary, YOLOv8n-IRP shows excellent detection and positioning performance, which not only provides a new method for the key point localization of the rubber-tapping robot but also provides technical support for the unmanned rubber-tapping operation of the intelligent rubber-tapping robot.</p>
</abstract>
<kwd-group>
<kwd>tapping surface detection</kwd>
<kwd>key point positioning</kwd>
<kwd>intelligent rubber-tapping robot</kwd>
<kwd>receptive-field attention</kwd>
<kwd>AFPN</kwd>
</kwd-group>
<counts>
<fig-count count="15"/>
<table-count count="5"/>
<equation-count count="11"/>
<ref-count count="36"/>
<page-count count="17"/>
<word-count count="8945"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>As the only renewable industrial raw material and strategic resource, natural rubber is often categorized as one of the four major industrial raw materials, along with steel, petroleum, and coal. Due to the unique physical properties of natural rubber: resilience, elasticity, abrasion resistance, impact resistance, efficient heat dissipation, and flexibility at low temperatures that cannot be replaced by synthetic alternatives, it is widely used in more than 50,000 products, such as aircraft tires, sporting goods, medical and scientific instruments, and insulated cables, which has led to a significant increase in the annual demand for natural rubber (<xref ref-type="bibr" rid="B24">Tan et&#xa0;al., 2023</xref>). According to the statistical report of the Rubber Research Institute of the Chinese Academy of Tropical Agricultural Sciences, the global natural rubber production in 2023 is 14.319 million tons, up 0.5%. The natural rubber consumption is 15.19 million tons, an increase of 0.8%. The global natural rubber production is forecast to reach 14.542 million tons in 2024, up 1.6%. The consumption is predicted to reach 15.67 million tons, an increase of 3.0%. At present, natural rubber tapping is mainly used to tap rubber by hand, and the commonly used rubber tapping tools are traditional tapping knives, handheld electric tapping knives, etc (<xref ref-type="bibr" rid="B2">Arjun et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B22">Soumya et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B35">Zhou et&#xa0;al., 2021</xref>). A rubber-tapping worker needs to tap more than 500 rubber trees per day, which is labor-intensive and requires high skills. However, rubber trees are mainly planted in the developing countries of Asia and South America, affected by the economic situation, regional politics, environmental climate, and many other factors, resulting in the price of natural rubber never a steady increase and even some decline. A severe blow to the motivation of workers resulted in the loss of many skilled workers and large areas of rubber forests facing abandonment, so the natural rubber industry is facing a labor shortage and an aging bottleneck (<xref ref-type="bibr" rid="B36">Zhou et&#xa0;al., 2022</xref>). Therefore, there is an urgent need to develop an intelligent rubber-tapping machine to reduce the work intensity of rubber workers, increase rubber-tapping yield, and solve the predicament of the natural rubber industry (<xref ref-type="bibr" rid="B36">Zhou et&#xa0;al., 2022</xref>). Among them, using machine vision to detect the tapping area and locate the starting and ending point of tapping is the key to realizing intelligent tapping. The rubber tapping area is composed of spiral lines tapped by rubber workers. The starting and end points of rubber tapping are located at the beginning and end of the spiral line. Whether the starting and end points of rubber tapping can be accurately positioned affects the quality and yield of rubber. However, during rubber tapping operations in rubber forests, complex factors such as uneven light exposure, different thicknesses of rubber trees of various ages, and unclear tapping line features make it difficult to locate the starting and ending points of tapping accurately.</p>
<p>In fact, the characteristics of the key points of tapping are small features located on the tapping surface. Therefore, whether the key points of tapping can be accurately located depends on whether the detailed features of the tapping surface can be fully extracted and whether the characteristics of the key points of tapping can be screened out from the numerous detailed features. This is similar to the problems encountered in most object detection tasks in the agricultural field, namely, how to extract object features and filter out important features. In recent years, with the development of machine vision and agricultural intelligence, machine vision has been widely applied in the agricultural field (<xref ref-type="bibr" rid="B20">Rehman et&#xa0;al., 2019</xref>), including the application of traditional machine learning methods and deep learning methods. In traditional machine learning methods, the object detection task is mainly performed by manually designed classifiers using the object&#x2019;s color, geometric, and texture features to classify and detect the object. For example, <xref ref-type="bibr" rid="B25">Tan et&#xa0;al. (2018)</xref> used the histogram of gradient direction and color features to distinguish blueberry fruits of different maturity. <xref ref-type="bibr" rid="B14">Lin et&#xa0;al. (2020)</xref> detected apricot varieties based on features of contour information. <xref ref-type="bibr" rid="B11">Li et&#xa0;al. (2016)</xref> combined color, shape and texture features to identify unripe green citrus fruits. The above methods have achieved certain results, but at the same time, they have also exposed some drawbacks. Traditional machine learning methods require a lot of time to perform manual feature selection and have limited adaptability in complex scenarios, which greatly hinders the performance and robustness of traditional machine learning methods for object detection in natural environments (<xref ref-type="bibr" rid="B4">Chen et&#xa0;al., 2024</xref>). This is extremely disadvantageous for detecting the tapping surface and locating the key points of rubber tapping in the complex rubber forest environment.</p>
<p>Deep learning is a powerful subcategory of machine learning. It can increase the depth and width of the entire large network through the continuous stacking of small modules, thereby improving the feature extraction capabilities of the network and having stronger feature extraction ability than traditional machine learning. At the same time, deep learning does not require manual feature selection and is highly adaptable to complex scenarios. Therefore, deep learning has become the preferred technology for identification and detection in the agricultural field (<xref ref-type="bibr" rid="B15">Liu and Liu, 2024</xref>; <xref ref-type="bibr" rid="B1">Altalak et&#xa0;al., 2022</xref>). So far, deep learning has been widely studied in many agrarian applications (<xref ref-type="bibr" rid="B27">Thakur et&#xa0;al., 2023</xref>), including weed detection (<xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B18">Ortatas et&#xa0;al., 2024</xref>), pest and disease detection (<xref ref-type="bibr" rid="B10">Kumar and Kukreja, 2022</xref>; <xref ref-type="bibr" rid="B26">Tang et&#xa0;al., 2024</xref>), fruit detection (<xref ref-type="bibr" rid="B29">Wang et&#xa0;al., 2023c</xref>; <xref ref-type="bibr" rid="B7">Guan et&#xa0;al., 2023</xref>), grain crop detection (<xref ref-type="bibr" rid="B21">Song et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B28">Wang et&#xa0;al., 2023b</xref>), and so on. Among them, the YOLO model, as a representative of the one-stage detection algorithm model, is slightly inferior to the two-stage detection algorithm models, such as Faster-RCNN and Mask-RCNN, in terms of detection accuracy, but its lightweight network structure design enables it to have a faster detection speed and a smaller model size. So, it has been widely used in various fields (<xref ref-type="bibr" rid="B3">Bello and Oladipo, 2024</xref>; <xref ref-type="bibr" rid="B30">Wang et&#xa0;al., 2023a</xref>; <xref ref-type="bibr" rid="B17">Mokayed et&#xa0;al., 2023</xref>). However, the two-stage detection algorithm model has a large number of parameters and requires greater computing power, which poses a challenge to the deployment of the model on the mobile terminal. In fact, the computing resources of the intelligent rubber-tapping robot are limited, and the detection speed will be seriously affected compared with the hardware configuration in the experimental environment. Therefore, the YOLO series model is more suitable for deployment in the rubber-tapping robot to realize intelligent rubber tapping.</p>
<p>At present, researchers have conducted little research on intelligent rubber tapping. <xref ref-type="bibr" rid="B23">Sun et&#xa0;al. (2022)</xref> proposed a natural rubber tree tapping trajectory detection method based on an improved YOLOv5 model, which realized the detection of the tapping surface and achieved a mAP50 of 95.1%. <xref ref-type="bibr" rid="B6">Chen et&#xa0;al. (2023)</xref> proposed a natural rubber tree tapping area detection and new tapping line positioning method based on an improved mask region convolutional neural network (Mask-RCNN), which realized the segmentation and extraction of tapping lines and located new tapping lines based on existing tapping lines, with the segmentation accuracy of tapping lines reaching 99.78%. The above scholars discussed the tapping surface and tapping line, respectively, but lacked research on the positioning of the starting and end points of tapping. Positioning the rubber-tapping starting point is the first step of the whole process. Without determining the position of the starting point of rubber tapping, the follow-up work of rubber tapping cannot be carried out. The accuracy of the positioning of the starting point of rubber tapping directly affects the quality of the glue flow after tapping. Positioning the end point of rubber tapping is the final step of the entire rubber tapping process, which involves the length of the tapping line. Currently, the commonly used secant lengths are 1/2 secant (the tapping surface is 1/2 of the rubber tree surface) and 1/4 secant (the tapping surface is 1/4 of the rubber tree surface). The efficiency of rubber flow is different for different tapping line lengths. Therefore, the accurate positioning of the starting and end points of tapping is of great significance in the whole tapping process. To this end, this paper proposes an improved YOLOv8n-IRP (Improved rubber tapping key point positioning) model based on YOLOv8n-Pose, which is used to detect the tapping surface of rubber trees and locate the starting and end points of rubber tapping. YOLOv8n-Pose is an end-to-end network that integrates object detection and key point detection. Its lightweight network structure makes it difficult for its detection and positioning accuracy in complex rubber forest environments to meet the actual rubber tapping requirements. To address this problem, this paper makes three improvements to the model. The main work and contributions are as follows:</p>
<list list-type="simple">
<list-item>
<p>(1) A data set of natural rubber tree tapping surface detection and starting point and end point positioning, including tapping surfaces of different tapping ages and tapping surfaces with different angles and light intensities, is established. Methods such as noise addition and picture splicing are used to preprocess the data set to improve the generalization ability and robustness of the model.</p>
</list-item>
<list-item>
<p>(2) The Receptive-field attention mechanism is integrated into the backbone network, which solves the problem of parameter sharing of larger convolution kernels in ordinary convolutions and calculates the importance of all features in the receptive field, thus improving the backbone network&#x2019;s feature extraction capability.</p>
</list-item>
<list-item>
<p>(3) The Asymptotic Feature Pyramid Network (AFPN) replaces the Path Aggregation Feature Pyramid Network (PAFPN) of the neck network, reducing the loss or degradation of high-level feature information in the top-down enhancement process and the loss and degradation of low-level feature information in the bottom-up enhancement process.</p>
</list-item>
<list-item>
<p>(4) A dual-branch key point detection head is designed based on the residual module. The dual-branch structure uses the sigmoid function as a gate to generate different weights for the two branches to screen out important features, while the residual structure makes up for important features lost during the feature screening process, enabling important features to be screened out as completely as possible.</p>
</list-item>
</list>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Data collection and annotation</title>
<p>The experiment is conducted on rubber trees tapped for one, three, and five year(s). In the National Natural Rubber Forest in Danzhou City, Hainan Province, China, 2029 photos are collected using image acquisition equipment, a Sony Alpha 6000 camera with a resolution of 4000&#xd7;6000. In order to ensure the richness and diversity of the samples, multi-angle shooting methods are used under different lighting conditions, and photos of 9 scenes are collected. It includes rubber trees with one, three, and five year(s) of tapping age. The rubber trees of each tapping age also include rubber trees that block the end point of tapping but not the starting point of tapping, rubber trees that block the starting point of tapping but not the end point of tapping, and rubber trees that both the starting point and end point are blocked at the same time, as shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. Finally, Labelme image annotation software is used to manually label the rubber tree&#x2019;s tapping area, tapping starting point, and tapping endpoint to create a JSON format data set.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Representative sample data set of different tapping ages.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g001.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Data enhancement</title>
<p>In deep learning network model training, the richness, diversity, and accuracy of the data set have a decisive impact on the final training results of the network model. The singleness and deficiency of the data set will lead to the model being overfitted. At the same time, due to the complex environment of the rubber forest, the use of machine vision to collect the tapping surface information of the rubber tree will be affected by unfavorable factors such as light and noise, which will lead to significant errors in the final identification and positioning. Therefore, it is necessary to enhance further the data set before network training to prevent over-fitting of the model and improve the generalization ability of the network model to adapt to the complex rubber forest environment. This study performed various random enhancement operations on the annotated original data set, including adding noise, changing light, changing pixels, translation, stitching multiple pictures, and flipping, as shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. In order to ensure the balance of the proportions of various categories in the data set, a method of different enhancement times for other categories is adopted. Categories with a smaller proportion have an increase in times of enhancement, while categories with a larger proportion have a reduced number of improvements. Finally, it is divided into a training set, a verification set, and a test set in a ratio of 8:1:1. The number of pictures is 5712, 715, and 715, respectively. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> shows the change in the number of category labels before and after the enhancement.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Sample data enhancement.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g002.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>The number of category labels before and after data augmentation.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Category</th>
<th valign="middle" align="center">Original</th>
<th valign="middle" align="center">Data Enhancement</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">starting-point</td>
<td valign="middle" align="center">1053</td>
<td valign="middle" align="center">2106</td>
</tr>
<tr>
<td valign="middle" align="center">ending-point</td>
<td valign="middle" align="center">244</td>
<td valign="middle" align="center">1220</td>
</tr>
<tr>
<td valign="middle" align="center">non-point</td>
<td valign="middle" align="center">732</td>
<td valign="middle" align="center">2196</td>
</tr>
<tr>
<td valign="middle" align="center">Mixed of three categories</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">1620</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Standard YOLOv8 network structure</title>
<p>YOLO (You Only Look Once) is the beginning of the One-Stage detection algorithm. Compared with Two-Stage algorithms, YOLO can greatly improve the detection speed while ensuring good detection accuracy. According to the scale of the network, the YOLOv8 model can be divided into five versions, namely YOLOv8n, YOLOv8s, YOLOv8m, YOLOv8l, and YOLOv8x, and each version includes three versions of object detection, segmentation, and key point detection. Considering the actual rubber-tapping situation, this article selected the lightweight YOLOv8n key point detection algorithm for research. Compared with the other four, YOLOv8n has a lightweight parameter structure, which is more conducive to deployment in small mobile devices.</p>
<p>The key point detection network structure of YOLOv8 is composed of a backbone network, a neck network, and a head network, as shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. First, the input image enters the backbone network within which the CBS module, C2F module, and SPPF module are used to extract features at various scales. Then, the neck network uses the Path Aggregation Feature Pyramid Network (PAFPN) structure to process further and fuse the extracted multi-scale features. Finally, the head network processes the fused feature maps at different levels to output the detection results.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>The architecture of Standard YOLOv8 Key point detection model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g003.tif"/>
</fig>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Improved YOLOv8n-IRP network structure</title>
<sec id="s3_1">
<label>3.1</label>
<title>Enhancement of backbone network feature extraction capabilities</title>
<p>Traditional convolution uses the same parameters in each receptive field to extract feature information through the convolution kernel without considering the different information between different positions. This results in a large amount of redundant information in the extracted data, which reduces the extraction time. The efficiency of features greatly limits the performance of the model. The emergence of the spatial attention mechanism enables the model to focus on certain key features (<xref ref-type="bibr" rid="B19">Park et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B12">Li et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B16">Luo et&#xa0;al., 2022</xref>), enhancing the network&#x2019;s ability to capture detailed feature information. However, it can only be used to solve the identification of spatial features and does not completely solve the parameter-sharing problem of larger convolution kernels (such as 3&#xd7;3 convolution). In addition, they cannot judge the importance of each feature in the receptive field, such as the existing Convolutional Block Attention Module (CBAM) (<xref ref-type="bibr" rid="B31">Woo et&#xa0;al., 2018</xref>) and Coordinate Attention(CA) (<xref ref-type="bibr" rid="B9">Hou et&#xa0;al., 2021</xref>).</p>
<p>The proposal of RFA solves the limitations of the existing spatial attention mechanism and provides an innovative solution for spatial processing. Among them, the Receptive-Field Attention Convolution (RFAConv) (<xref ref-type="bibr" rid="B33">Zhang et&#xa0;al., 2023a</xref>) designed based on RFA not only emphasizes the importance of different features within the receptive field slider but also gives priority to the receptive field space features, completely solving the problem of convolution kernel parameter sharing, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>The overall structure of RFAConv. S, Softmax.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g004.tif"/>
</fig>
<p>In RFA, the entire operation process can be divided into two parts. The first part uses group convolution to extract receptive field spatial features quickly. The second part learns the attention map by interacting with the receptive field feature information to enhance the network&#x2019;s ability to extract features. However, allowing each receptive field feature to interact will incur a large computational cost. To reduce the computational cost and parameter amount as much as possible, AvgPool is first used to fuse the global information of each receptive field feature, followed by a 1&#xd7;1 group convolution operation to interact with the information. Finally, the Softmax function obtains the importance of each feature in the receptive field feature. After both parts are completed, the final feature information is obtained by multiplication, as shown in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>.</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mi>F</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the attention map and the transformed receptive field space feature map, respectively; <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are group convolutions of size <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, respectively; <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>X</italic> are normalization and input features, respectively.</p>
<p>The feature map obtained through RFA will not overlap the receptive fields after shape adjustment. Therefore, the learned attention map not only contains all the feature information in each receptive field but does not need to be shared in each receptive field. Finally, a standard convolution with a convolution kernel of <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and a stride of <italic>k</italic> is used to extract feature information.</p>
<p>Consequently, in this paper, by replacing the standard convolutional Conv in the CBS module of the backbone network with RFAConv as depicted in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, the feature extraction capability of the backbone network is improved, while the increase in the computational cost and the number of parameters are almost negligible.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Comparison chart before and after CBS module improvement.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g005.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Mitigation of neck network feature loss and degradation</title>
<p>In YOLOv8, the main task of the backbone network is feature extraction, but in detection and positioning tasks, the detected objects are multi-scale, and single-scale features cannot be used to detect multi-scale objects. Therefore, YOLOv8 uses the PAFPN structure in the neck network to process the features extracted from the backbone. Initially, the features are fused from top to bottom and then enhanced from bottom to top before generating a multi-scale feature map. Nonetheless, this approach encounters a new issue. In the process of top-down fusion, the high-level feature information may be lost or degraded, while in the bottom-up process, the low-level feature information may be lost or degraded. To address this problem, this paper references the Asymptotic Feature Pyramid Network (AFPN) (<xref ref-type="bibr" rid="B32">Yang et&#xa0;al., 2023</xref>) in the neck network, as shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, to replace the original PAFPN.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>The architecture of AFPN.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g006.tif"/>
</fig>
<p>As seen in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, AFPN sequentially fuses the feature information of the bottom, middle, and top layers. This process is carried out gradually, which greatly alleviates the problem of poor feature fusion effect caused by excessive feature differences between non-adjacent layers. For example, feature fusion between the low and middle layers reduces the feature difference between them. Since the middle and high layers are adjacent layers, the feature differences between the low and high layers are also reduced.</p>
<p>The main task of the ASFF module in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> is to assign different spatial weights to features at various levels in the multi-level feature fusion process, which enhances the importance of key levels and reduces the impact of conflicting information between different levels. In this article, the ASFF module is divided into two modes, including ASFF2 and ASFF3. Among them, ASSF2_1 and ASSF2_2 denote level 2 feature fusion with two different weights, while ASSF3_1, ASSF3_2, and ASSF3_3 denote level 3 feature fusion with three different weights. Taking level 3 feature fusion as an example, the operation process is as follows:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msubsup>
<mml:mtext>y</mml:mtext>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the feature vector at position <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> from level <italic>n</italic> to level <italic>l</italic>; <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are the three spatial weights at level <italic>l</italic>, and the constraint is <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>; <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the feature obtained after the final fusion.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Improvement of head network</title>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Design of key point detection module</title>
<p>Deep networks extract low-level, mid-level, and high-level features of the input in an end-to-end manner. The richness of feature extraction affects the detection and classification accuracy in the later stages of the network. The network can learn richer features through the number of stacked layers (<xref ref-type="bibr" rid="B8">He et&#xa0;al., 2016</xref>), thereby improving detection and classification accuracy. However, as the number of layers (depth) continues to increase, the improvement of network detection and classification accuracy is not absolute. Because each layer of the network also loses part of the feature information while extracting features, the lost features may include some important features, while the extracted features may only be some secondary features and not important features. So, although the number of layers has increased, and the extracted features have become richer, they are likely to be some useless features. Not only will the accuracy not be improved, but the accuracy will be reduced. At the same time, new problems will appear in the network; for example, the gradient may disappear or explode, making the network unable to converge.</p>
<p>Therefore, based on &#x200b;&#x200b;the residual network (<xref ref-type="bibr" rid="B8">He et&#xa0;al., 2016</xref>), this paper designs a dual-branch key point detection module, as shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Improved key point detection head.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g007.tif"/>
</fig>
<p>Compared with the standard key point detection head shown in <xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3</bold>
</xref>, <xref ref-type="fig" rid="f7">
<bold>7</bold>
</xref> consists of four standard points, and two of them are connected in parallel to form a new structure, as shown in the red box in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. In this new structure, a sigmoid function is added to one of the columns to generate a weight value between (0-1). Two identical new structures are connected in parallel, each extracting different features. Then, the importance of the extracted features in the entire module is determined by their respective weight values, w. Finally, the original input X is added to compensate for losing important feature information.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Elimination of redundant features</title>
<p>As the network structure becomes more and more complex, some convolutional layers will extract redundant features, resulting in a huge waste of computing resources. In order to reduce redundant calculations and promote the learning of representative features, this paper adds the Spatial and Channel reconstruction Convolution (SCConv) (<xref ref-type="bibr" rid="B13">Li et&#xa0;al., 2023</xref>) to the designed dual-branch key point detection module. SCConv consists of two units: spatial reconstruction unit (SRU) and channel reconstruction unit (CRU), as shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>. The SRU uses a split-reconstruction method to suppress spatial redundancy, while the CRU employs a split-transform-fusion strategy to reduce channel redundancy.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>The architecture of SCConv. GN: Group Normalization. N: <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>. S, Sigmoid; C, Concatenation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g008.tif"/>
</fig>
<p>The SRU consists of two parts: separation operation and reconstruction operation. In the separation operation, the input feature map <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> (<italic>N</italic>, <italic>C</italic>, <italic>H</italic>, and <italic>W</italic> are training batch, number of channels, height, and width, respectively) is first standardized to obtain the trainable parameter <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mi>C</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, as shown in <xref ref-type="disp-formula" rid="eq3">Equation 3</xref>. Then, <inline-formula>
<mml:math display="inline" id="im19">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> is normalized to obtain the relevant weight <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mi>C</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and the weight <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is mapped to (0, 1) using the sigmoid function to indicate the importance of different feature maps, as shown in <xref ref-type="disp-formula" rid="eq4">Equation 4</xref>. Finally, the threshold is used for gating to obtain weights <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, while the input feature map <italic>X</italic> is multiplied with it to obtain <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>w</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> with rich information and <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>w</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> with less information, thus realizing the separation of feature maps with rich information and feature maps with less spatial content, as shown in <xref ref-type="disp-formula" rid="eq5">Equation 5</xref>.</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mtext>out</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>G</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>{</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>}</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>G</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>G</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <inline-formula>
<mml:math display="inline" id="im26">
<mml:mi>&#x3bc;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im27">
<mml:mi>&#x3c3;</mml:mi>
</mml:math>
</inline-formula> are the mean and standard deviation of <italic>X</italic>; <inline-formula>
<mml:math display="inline" id="im28">
<mml:mi>&#x3f5;</mml:mi>
</mml:math>
</inline-formula> is a small positive number added for division stability; <inline-formula>
<mml:math display="inline" id="im29">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im30">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula> are trainable affine transformations; <inline-formula>
<mml:math display="inline" id="im31">
<mml:mo>&#x2297;</mml:mo>
</mml:math>
</inline-formula> is an element-wise multiplication.</p>
<p>To maintain the information flow between feature information, the reconstruction operation is used after the separation operation to fully combine the two different information features, so as to enhance the important features and suppress the redundant features in the spatial dimension, and finally obtain the Spatial-Refined Feature Maps <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, as shown in <xref ref-type="disp-formula" rid="eq6">Equation 6</xref>.</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mn>11</mml:mn>
</mml:mrow>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mo>&#x2295;</mml:mo>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mn>21</mml:mn>
</mml:mrow>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mo>&#x2295;</mml:mo>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mn>12</mml:mn>
</mml:mrow>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x222a;</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <inline-formula>
<mml:math display="inline" id="im33">
<mml:mo>&#x2295;</mml:mo>
</mml:math>
</inline-formula> is an element-wise summation, and &#x222a; is the Concatenation operation.</p>
<p>After applying SUR to the intermediate input feature <italic>X</italic>, although the redundant features in the spatial dimension can be suppressed, the redundancy in the channel dimension is still maintained, which is caused by the repeated use of standard convolution with a convolution kernel of <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Therefore, in order to eliminate channel redundancy, the channel reconstruction unit (CRU) is introduced to replace the standard convolution.</p>
<p>The CRU consists of three parts: segmentation, transformation, and fusion. First, CRU performs channel segmentation on Spatial-Refined Feature Maps <inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and uses 1&#xd7;1 convolution to compress the two feature maps obtained after segmentation to improve computational efficiency, and obtains the upper feature <inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the lower feature <inline-formula>
<mml:math display="inline" id="im37">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively. Then, <inline-formula>
<mml:math display="inline" id="im38">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with rich features is sent to the upper transformer, as shown in <xref ref-type="disp-formula" rid="eq7">Equation 7</xref>, and <inline-formula>
<mml:math display="inline" id="im39">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with a large number of redundant features is sent to the lower transformer, as shown in <xref ref-type="disp-formula" rid="eq8">Equation 8</xref>. Finally, the simplified SKNet method is used to adaptively merge the output features <inline-formula>
<mml:math display="inline" id="im40">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im41">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the upper transformer and the lower transformer, so that the redundancy in the channel dimension is suppressed, and the channel-refined features <italic>Y</italic> is obtained, as shown in <xref ref-type="disp-formula" rid="eq9">Equation 9</xref>.</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x222a;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>H</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>W</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mstyle>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mi>G</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>r</mml:mi>
</mml:mfrac>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are the learnable weight matrices of GWC and PWC, respectively; <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mtext>up</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>r</mml:mi>
</mml:mfrac>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are the upper layer input and output feature maps, respectively; <inline-formula>
<mml:math display="inline" id="im46">
<mml:mrow>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>r</mml:mi>
</mml:mfrac>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mi>r</mml:mi>
</mml:mfrac>
</mml:mstyle>
<mml:mo stretchy="false">)</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the learnable matrix of PWC; &#x222a; is the Concatenation operation; <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>r</mml:mi>
</mml:mfrac>
</mml:mstyle>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are the lower layer input and output feature maps, respectively.</p>
</sec>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Model evaluation indicators</title>
<p>This paper evaluates the comprehensive performance of the model through two parts of experiments. The first part of the experiment: Rubber tree tapping surface detection and rubber tapping key point detection accuracy experiments, using Precision (<italic>P</italic>), Recall (<italic>R</italic>), Mean Average Precision (<italic>mAP</italic>), model parameters (Params), Flops, and FPS as evaluation indicators. Among them, <italic>P</italic> and <italic>R</italic> represent the proportion of the number of correctly predicted positive samples to the total number of predicted positive samples and the proportion of the number of correctly predicted positive samples to all positive samples, respectively; <italic>mAP</italic> is the average area under the P-R curve of all categories, which is used to measure the quality of the model in each category, among which <italic>mAP50</italic> is the <italic>mAP</italic> value when the IOU threshold is set to 0.5; FLOPs and FPS respectively show the computing power required for model training and the inference speed of the model (the number of images inferred in 1 second).</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mi>d</mml:mi>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <italic>TP</italic>, <italic>FP</italic>, and <italic>FN</italic> represent the number of samples correctly predicted by the model as positive (i.e., the target exists and is predicted to exist), the number of samples incorrectly predicted by the model as positive (i.e., the target does not exist but is predicted to exist), and the number of samples incorrectly predicted by the model as negative (i.e., the target exists but is predicted to not exist); <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>N</italic> show the precision, recall and number of sample categories, respectively.</p>
<p>The second part of the experiment: Experiment on the positioning accuracy of the starting point and end point of rubber tapping, using <italic>x</italic>-axis offset distance (<italic>xOD</italic>), <italic>y</italic>-axis offset distance (<italic>yOD</italic>), and <italic>xy</italic>-axis offset distance (<italic>xyOD</italic>) as the evaluation indexes. <italic>xOD</italic>, <italic>yOD</italic>, and <italic>xyOD</italic> represent the pixel offset distances between the predicted point and the truth point in the <italic>x</italic>-axis direction, the <italic>y</italic>-axis direction, and the Euclidean direction, respectively. The calculation formulas are as follows:</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>O</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mi>O</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>O</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im52">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im53">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im54">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the <italic>x</italic>-axis coordinates and <italic>y</italic>-axis coordinates of the predicted point, and the <italic>x</italic>-axis coordinates and <italic>y</italic>-axis coordinates of the truth point, respectively.</p>
</sec>
</sec>
<sec id="s4" sec-type="results">
<label>4</label>
<title>Results and discussion</title>
<sec id="s4_1">
<label>4.1</label>
<title>Ablation experiment</title>
<p>The natural rubber tree tapping surface detection and tapping key point positioning model has been improved in three parts compared to the original YOLOv8n-pose model. Part I A: The convolution RFAConv with receptive field attention mechanism replaces the ordinary convolution in the backbone network CBS module; Part II B: Neck network uses AFPN structure; Part III C: An improved key point detection head is adopted in the Head network. To verify the contribution of each improvement to the entire model, this study conducts an Ablation experiment on the natural rubber tree tapping surface detection and tapping key point positioning model. The results are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison results of ablation experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">A</th>
<th valign="middle" align="center">B</th>
<th valign="middle" align="center">C</th>
<th valign="middle" align="center">D_P/P_P (%)</th>
<th valign="middle" align="center">D_R/P_R (%)</th>
<th valign="middle" align="center">D_mAP50/P_mAP50 (%)</th>
<th valign="middle" align="center">Params<break/>(M)</th>
<th valign="middle" align="center">GFlops<break/>(G)</th>
<th valign="middle" align="center">FPS<break/>(f/s)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">96.3/86.2</td>
<td valign="middle" align="center">97.9/87.6</td>
<td valign="middle" align="center">96.9/84.1</td>
<td valign="middle" align="center">3.08</td>
<td valign="middle" align="center">8.3</td>
<td valign="middle" align="center">200</td>
</tr>
<tr>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">97.7/87.8</td>
<td valign="middle" align="center">99.1/89.7</td>
<td valign="middle" align="center">97.8/85.6</td>
<td valign="middle" align="center">3.10</td>
<td valign="middle" align="center">8.6</td>
<td valign="middle" align="center">167</td>
</tr>
<tr>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">98.2/89.1</td>
<td valign="middle" align="center">98.7/89.6</td>
<td valign="middle" align="center">97.6/85.3</td>
<td valign="middle" align="center">3.21</td>
<td valign="middle" align="center">9.6</td>
<td valign="middle" align="center">111</td>
</tr>
<tr>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">96.4/87.3</td>
<td valign="middle" align="center">97.9/88.4</td>
<td valign="middle" align="center">96.8/86.2</td>
<td valign="middle" align="center">3.16</td>
<td valign="middle" align="center">8.6</td>
<td valign="middle" align="center">143</td>
</tr>
<tr>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">98.5/88.9</td>
<td valign="middle" align="center">99.2/89.8</td>
<td valign="middle" align="center">98.3/86.4</td>
<td valign="middle" align="center">3.31</td>
<td valign="middle" align="center">10.1</td>
<td valign="middle" align="center">91</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>
<sup>1)</sup> D_P, D_R, D_mAP50 and P_P, P_R, P_mAP50 denote P, R, and mAP50 for object detection and key point detection, respectively.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The integration of the receptive field attention mechanism has comprehensively improved the P, R, and mAP50 of the object detection and key point detection of the original model. As shown in the experimental results of YOLOv8n-Pose+A, D_P, D_R, and D_mAP50 are improved by 1.4%, 1.2%, and 0.9%, respectively, while P_P, P_R, and P_mAP50 are improved by 1.6%, 2.1%, and 1.5%, respectively, indicating that the feature extraction capability of the backbone network has been enhanced. Meanwhile, the number of parameters and computing power cost has only increased by 0.02M and 0.3G, respectively, further proving that the receptive field attention mechanism has little impact on the size and computing cost of the entire model. After replacing the original PAFPN structure of the Neck network with AFPN, the problem of loss and degradation of bottom-level feature information and top-level feature information has been alleviated. Compared with the original model, the D_P and P_P of the YOLOv8n-Pose+B model are substantially improved by 1.9% and 2.9%, respectively. However, due to the operation of progressive fusion, feature fusion becomes more frequent, which in turn generates more parameters and computing power, increasing by 0.13M and 1.3G, respectively. After designing the original single-branch key point detection head of the Head network into a dual-branch key point detection head and introducing the residual structure, the detection head&#x2019;s ability to select important features of key points is effectively enhanced. At the same time, the residual structure further compensates for the loss of important features. The P_mAP50 is improved by 2.1% compared with the original model. The combination of RFAConv, AFPN, and enhanced key point detection head showed the best detection performance, with D_mAP50 and P_mAP50 increased by 1.4% and 2.3%, respectively, compared with the original model. Still, it also increased the model complexity, increasing model size and computing power and a slower model inference speed. The biggest impact is the inference speed of the model. Although the FPS dropped to 91, the actual tapping time is 45s, and the tapping speed is about 0.8cm/s. Therefore, the detection speed of 91FPS fully meets the requirements of real-time tapping. The number of model parameters and computing power has only increased slightly, with Params increased to 3.31M and GFlops increased to 10.1G. In the current application of lightweight models in agricultural fields, <xref ref-type="bibr" rid="B23">Sun et&#xa0;al. (2022)</xref> proposed a lightweight model with 6.84M Params and 14.7G GFlops for rubber tapping, and <xref ref-type="bibr" rid="B34">Zhang et&#xa0;al. (2023b)</xref> proposed a lightweight model with 4.78M Params and 12.3G GFlops for animal recognition. In comparison, YOLOv8n-IRP is much smaller than them in parameters and GFlops, which is very beneficial for deployment on intelligent rubber-tapping machines.</p>
<p>
<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> shows the training results of the model more intuitively. Before 30 epochs, the loss of the object and key points decreases rapidly, while the mAP50 increases rapidly, indicating that the model has a faster convergence rate both before and after the improvement, and does not decrease due to the increase in model complexity. Simultaneously, the improved YOLOv8n-IRP has lower loss and higher mAP50 than the original model. This observation shows that the combination of RFAConv, AFPN, and the improved key point detection head enables the model to have better detection performance.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Comparison of loss and mAP50 curves in ablation experiments. <bold>(A, B)</bold> The convergence of the object loss and the D_mAP50. <bold>(C, D)</bold> The convergence of the key point loss and the P_mAP50.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g009.tif"/>
</fig>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Comparison of detection performance between different models</title>
<p>To demonstrate the comprehensive performance of the improved YOLOv8n-IRP model in rubber tree tapping surface detection and rubber tapping key point detection, this experiment uses three popular object detection and key point detection algorithms for comparison, as shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. In <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, except for Faster_RCNN-RTMPose, the other three algorithms are lightweight models. Among them, the improved lightweight model YOLOv8-IRP has the highest D_P and D_R, P_P, P_R, D_mAP50 and P_mAP50, which are second only to the Faster_RCNN-RTMPose, reaching 98.5%, 88.9%, 99.2%, 89.8%, 98.3% and 86.4%, respectively. The reason why the Faster_RCNN-RTMPose can show high detection accuracy in key point detection is due to the detection mode of RTMPose. The RTMPose is a top-down key point detection algorithm. It first detects the object box and then predicts the key points in the object box by generating a key point heat map. This makes detection accuracy better than lightweight models that simultaneously predict objects and key points through regression. However, it also exposes its shortcomings. It needs to train two models: the object detection model Faster_RCNN and the key point detection model RTMPose, which makes its model size larger and requires higher computing power to train the model. The increase in model size and the cumbersome detection steps also greatly reduce the detection speed. As shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, the Faster_RCNN-RTMPose has the largest Params and GFlops, reaching 54.42M and 199.5G, respectively, and the smallest FPS, only 13f/s, which is extremely disadvantageous for deployment on mobile devices. On the other hand, the Params and GFlops of the YOLOv8n-IRP have only 3.31M and 10.1G, which are dozens of times smaller than the Faster_RCNN-RTMPose. At the same time, the FPS can reach 91f/s, which is several times faster than the Faster_RCNN-RTMPose, so it is more suitable for deployment on mobile devices.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison results of detection performance of different network models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">D_P/P_P (%)</th>
<th valign="middle" align="center">D_R/P_R (%)</th>
<th valign="middle" align="center">D_mAP50/P_mAP50 (%)</th>
<th valign="middle" align="center">Params<break/>(M)</th>
<th valign="middle" align="center">GFlops<break/>(G)</th>
<th valign="middle" align="center">FPS<break/>(f/s)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Faster_RCNN-RTMPose</td>
<td valign="middle" align="center">96.5/96.6</td>
<td valign="middle" align="center">99.7/97.9</td>
<td valign="middle" align="center">98.7/93.0</td>
<td valign="middle" align="center">54.42</td>
<td valign="middle" align="center">199.5</td>
<td valign="middle" align="center">13</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv5n-Pose</td>
<td valign="middle" align="center">97.1/86.7</td>
<td valign="middle" align="center">98.2/87.8</td>
<td valign="middle" align="center">96.4/83.6</td>
<td valign="middle" align="center">2.58</td>
<td valign="middle" align="center">7.3</td>
<td valign="middle" align="center">167</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8n-Pose</td>
<td valign="middle" align="center">96.3/86.2</td>
<td valign="middle" align="center">97.9/87.6</td>
<td valign="middle" align="center">96.9/84.1</td>
<td valign="middle" align="center">3.08</td>
<td valign="middle" align="center">8.3</td>
<td valign="middle" align="center">200</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8n-IRP</td>
<td valign="middle" align="center">98.5/88.9</td>
<td valign="middle" align="center">99.2/89.8</td>
<td valign="middle" align="center">98.3/86.4</td>
<td valign="middle" align="center">3.31</td>
<td valign="middle" align="center">10.1</td>
<td valign="middle" align="center">91</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The rubber forest mainly includes rubber trees with one, three, and five year(s) of tapping age. Therefore, this experiment visualizes the detection results of four models for rubber trees with one, three, and five years of harvesting age, as shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>. Among them, YOLOv8-IRP achieves more than 96% confidence in the detection of tapping surfaces at one, three, and five year(s), which is 2-3% higher than YOLOv5n-Pose and YOLOv8n-Pose, and it can accurately detect the presence of the starting and end point. Although compared with the Faster_RCNN-RTMPose, it fails to predict the occluded key points (the occluded key points predicted by the Faster_RCNN-RTMPose are shown in the green dotted circles in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>), in actual rubber tapping, rubber tapping can only be carried out if the tapping key points are revealed. The occluded starting and ending points have no practical significance for rubber tapping. Therefore, the detection accuracy of the YOLOv8n-IRP meets the requirements of rubber tapping. In addition, The YOLOv5n-Pose and YOLOv8n-Pose have false detection in key point detection, which is mainly manifested in detecting key points from the tapping surface without key points, as shown in the yellow dotted circle in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>. This is because the tapping surfaces that expose key points have similar features to those that do not, while the Neck network structure of both YOLOv5n-Pose and YOLOv8n-Pose is PAFPN. The loss or degradation of low-level and high-level features will occur during the feature fusion process. Therefore, it is not possible to distinguish such similar features well enough to make correct predictions. To this end, this paper first uses the RFAConv in the YOLOv8n-IRP to enhance the ability of feature extraction. Then, it uses the AFPN structure to reduce the loss and degradation of low-level and high-level features in the feature fusion process. Finally, the designed dual-branch key point detection head is used to improve the feature screening ability and solve the problem of low prediction accuracy of similar features.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>The detection results of different tapping ages trees. Letters <bold>(A&#x2013;D)</bold> represent the detection results of the Faster_RCNN-RTMPose, YOLOv5n-Pose, YOLOv8n-Pose, and YOLOv8n-IRP, respectively. Numbers 1, 2, and 3 denote rubber trees with one, three, and five year(s) of tapping age.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g010.tif"/>
</fig>
<p>In addition to this, the uncertainty of weather and the shading of light by rubber tree trunk foliage result in variable lighting, which is one of the main challenges for vision applications in rubber forests. Therefore, in order to further demonstrate the usefulness of the improved YOLOv8n-IRP model in rubber forests, this experiment is conducted to test the overexposed, underexposed and normally exposed pictures using four models, respectively, and the detection success rates of the four models in the face of different lighting conditions are counted, as shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. In <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, the overexposed, underexposed and normal exposure images used for testing are 200 images, respectively, in which the YOLOv8n-IRP model achieves a detection success rate of 91% in the normal exposure environment, which is more than 5% higher compared to both YOLOv5n-Pose and YOLOv8n-Pose, achieving a higher detection accuracy. For overexposure and underexposure, the detection success rates of the four models have decreased to different degrees, which is caused by 1) the insufficient number of images of complex scenes in the training set and 2) the increased difficulty of extracting important features in complex scenes, which makes the models suffer from the phenomena of misdetection and underdetection. Although the accuracy of YOLOv8n-IRP is reduced by the influence of complex illumination conditions, it still maintains an average detection success rate of 87%, which significantly improves the detection accuracy compared with the original model YOLOv8n-Pose. As shown in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>, the duplicate detection and misdetection that originally appeared in overexposure and underexposure are improved, which indicates that YOLOv8n-IRP has a more excellent feature extraction capability and enhanced robustness. While Faster_RCNN-RTMPose has a slightly higher detection accuracy than YOLOv8n-IRP in various exposure scenarios, YOLOv8n-IRP is more suitable to be deployed in mobile devices for intelligent rubber tapping use, considering the detection accuracy, model size, detection speed and the actual situation of rubber tapping.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Detection results of different models under different lighting conditions.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Light intensity</th>
<th valign="middle" align="center">NSD</th>
<th valign="middle" align="center">NFD</th>
<th valign="middle" align="center">DSR (%)</th>
<th valign="top" align="center">ADSR (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="3" align="center">Faster_RCNN-RTMPose</td>
<td valign="middle" align="center">overexposed</td>
<td valign="middle" align="center">169</td>
<td valign="middle" align="center">31</td>
<td valign="middle" align="center">84.5</td>
<td valign="middle" rowspan="3" align="center">89.5</td>
</tr>
<tr>
<td valign="middle" align="center">underexposed</td>
<td valign="middle" align="center">180</td>
<td valign="middle" align="center">20</td>
<td valign="middle" align="center">90</td>
</tr>
<tr>
<td valign="middle" align="center">normal exposure</td>
<td valign="middle" align="center">188</td>
<td valign="middle" align="center">12</td>
<td valign="middle" align="center">94</td>
</tr>
<tr>
<td valign="middle" rowspan="3" align="center">YOLOv5n-Pose</td>
<td valign="middle" align="center">overexposed</td>
<td valign="middle" align="center">147</td>
<td valign="middle" align="center">53</td>
<td valign="middle" align="center">73.5</td>
<td valign="middle" rowspan="3" align="center">79</td>
</tr>
<tr>
<td valign="middle" align="center">underexposed</td>
<td valign="middle" align="center">160</td>
<td valign="middle" align="center">40</td>
<td valign="middle" align="center">80</td>
</tr>
<tr>
<td valign="middle" align="center">normal exposure</td>
<td valign="middle" align="center">167</td>
<td valign="middle" align="center">33</td>
<td valign="middle" align="center">83.5</td>
</tr>
<tr>
<td valign="middle" rowspan="3" align="center">YOLOv8n-Pose</td>
<td valign="middle" align="center">overexposed</td>
<td valign="middle" align="center">154</td>
<td valign="middle" align="center">46</td>
<td valign="middle" align="center">77</td>
<td valign="middle" rowspan="3" align="center">80</td>
</tr>
<tr>
<td valign="middle" align="center">underexposed</td>
<td valign="middle" align="center">157</td>
<td valign="middle" align="center">43</td>
<td valign="middle" align="center">78.5</td>
</tr>
<tr>
<td valign="middle" align="center">normal exposure</td>
<td valign="middle" align="center">169</td>
<td valign="middle" align="center">31</td>
<td valign="middle" align="center">84.5</td>
</tr>
<tr>
<td valign="middle" rowspan="3" align="center">YOLOv8n-IRP</td>
<td valign="middle" align="center">overexposed</td>
<td valign="middle" align="center">168</td>
<td valign="middle" align="center">32</td>
<td valign="middle" align="center">84</td>
<td valign="middle" rowspan="3" align="center">87</td>
</tr>
<tr>
<td valign="middle" align="center">underexposed</td>
<td valign="middle" align="center">171</td>
<td valign="middle" align="center">29</td>
<td valign="middle" align="center">85.5</td>
</tr>
<tr>
<td valign="middle" align="center">normal exposure</td>
<td valign="middle" align="center">182</td>
<td valign="middle" align="center">18</td>
<td valign="middle" align="center">91</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>
<sup>1)</sup> NSD, Number of successful detections; NFD, Number of failed detections; DSR, Detection success rate; ADSR, Average detection success rate.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Comparison of detection results under different exposure environments before and after model improvement. Letters <bold>(A, B)</bold> indicate overexposure and underexposure environments, respectively. Numbers 1 and 2 denote the YOLOv8n-Pose and YOLOv8n-IRP models, respectively.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g011.tif"/>
</fig>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Key point positioning performance comparison experiment</title>
<p>To demonstrate the positioning accuracy of the improved YOLOv8n-IRP model at the starting and ending points of rubber tapping, this experiment calculates the <italic>xOD</italic>, <italic>yOD</italic>, and <italic>xyOD</italic> of 550 key points predicted by the four models, and their average values &#x200b;&#x200b;are shown in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> and <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>. As can be seen from <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12A</bold>
</xref>, the average error of the YOLOv8n-IRP on the <italic>x</italic>-axis and <italic>y</italic>-axis is lower than that of the YOLOv8n-Pose and YOLOv5n-Pose, and the accuracy has been significantly improved. It can be seen from <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> that the average offset error of the YOLOv8n-IRP in the x-axis direction is only 23.05 pixels, which is the smallest error among the four models; the average offset error in the y-axis and Euclidean directions is similar to that of the Faster_RCNN-RTMPose and is reduced by more than 10 pixels compared to YOLOv8n-Pose and YOLOv5n-Pose. From <xref ref-type="fig" rid="f12">
<bold>Figures&#xa0;12B, C</bold>
</xref>, it can be seen that the stability of the localization error of YOLOv8n-IRP is greatly improved compared with that of YOLOv8n-Pose, in which the maximum error does not exceed 100 pixels, while YOLOv8n-Pose shows an error of close to 180 pixels, further proving that the positioning accuracy is improved after the model improvement. Positioning accuracy and stability affects the regularity of the tapping surface, which in turn affects the efficiency of glue flow. Therefore, the improvement of YOLOv8n-IRP positioning performance has improved the efficiency of glue flow, thereby increasing the latex yield.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Experimental results of comparing the positioning accuracy of different models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">X-axis average offset(pixel)</th>
<th valign="middle" align="center">Y-axis average offset(pixel)</th>
<th valign="middle" align="center">Average Euclidean distance(pixel)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Faster_RCNN-RTMPose</td>
<td valign="middle" align="center">25.05</td>
<td valign="middle" align="center">21.80</td>
<td valign="middle" align="center">33.18</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv5n-Pose</td>
<td valign="middle" align="center">31.81</td>
<td valign="middle" align="center">36.56</td>
<td valign="middle" align="center">53.53</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8n-Pose</td>
<td valign="middle" align="center">28.62</td>
<td valign="middle" align="center">36.75</td>
<td valign="middle" align="center">51.41</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8n-IRP</td>
<td valign="middle" align="center">23.05</td>
<td valign="middle" align="center">25.67</td>
<td valign="middle" align="center">38.45</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Error distribution scatterplot. <bold>(A)</bold> The average error of the 550 key points predicted by each model in the x-axis and y-axis directions. <bold>(B)</bold> The error distribution of 550 key points predicted by the YOLOv8n-IRP. <bold>(C)</bold> The error distribution of 550 key points predicted by the YOLOv8n-Pose.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g012.tif"/>
</fig>
<p>To further prove the feasibility of key point positioning of the YOLOv8n-IRP model, this experiment visualizes four models&#x2019; key point positioning results for rubber trees of different tapping ages, as shown in <xref ref-type="fig" rid="f13">
<bold>Figures&#xa0;13</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f15">
<bold>15</bold>
</xref>. Among them, the key points predicted by the YOLOv8n-IRP on rubber trees with one, three, and five year(s) of tapping age are close to the truth key points and show high positioning stability, as shown in the red dotted box in <xref ref-type="fig" rid="f13">
<bold>Figures&#xa0;13</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f15">
<bold>15</bold>
</xref>. However, the positioning deviation of YOLOv8n-Pose and YOLOv5n-Pose are obvious, with large error fluctuations. The Faster_RCNN-RTMPose has the lowest average offset error in the y-axis and Euclidean direction among the four models. Still, it is only a few pixels lower than the improved YOLOv8n-IRP, which is a small improvement for a 4000&#xd7;6000 pixel photo. Nevertheless, in the visualization experiment, although Faster_RCNN-RTMPose achieves the highest positioning accuracy, as shown by the green dashed box in <xref ref-type="fig" rid="f13">
<bold>Figures&#xa0;13</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f15">
<bold>15</bold>
</xref>, there were also tapping surfaces with poor positioning, as shown by the yellow dashed box in <xref ref-type="fig" rid="f13">
<bold>Figures&#xa0;13</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f15">
<bold>15</bold>
</xref>, indicating that the positioning error of Faster_RCNN-RTMPose fluctuates greatly. To sum up, the YOLOv8n-IRP shows better performance in locating key points on the tapping surface, which better meets the rubber tapping requirements.</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>The positioning of key points on the tapping surface with one year of tapping age. Letters <bold>(A&#x2013;D)</bold> represent the detection results of the Faster_RCNN-RTMPose, YOLOv5n-Pose, YOLOv8n-Pose, and YOLOv8n-IRP, respectively. Numbers 1 and 2 denote tapping surfaces with starting and end points. The red dot is the predicted starting point of tapping. The pink dot is the predicted end point of tapping. The green dot is the key point of truth.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g013.tif"/>
</fig>
<fig id="f14" position="float">
<label>Figure&#xa0;14</label>
<caption>
<p>The positioning of key points on the tapping surface with three years of tapping age. Part labels have the same meaning as <xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g014.tif"/>
</fig>
<fig id="f15" position="float">
<label>Figure&#xa0;15</label>
<caption>
<p>The positioning of key points on the tapping surface with five years of tapping age. Part labels have the same meaning as <xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1468188-g015.tif"/>
</fig>
</sec>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions and future work</title>
<p>In this paper, a rubber tree tapping surface detection and rubber tapping key point localization model is proposed based on the YOLOv8n-Pose. Firstly, the Receptive-field attention mechanism is integrated into the backbone network to solve the problem of sharing common convolutional parameters with larger convolutional kernels, thus improving the feature extraction capability of the backbone network. Secondly, the AFPN is introduced to reduce the loss and degradation of the underlying feature information and the higher-level feature information in feature fusion and enhancement. Finally, a dual-branch key point detection head is designed based on the residual module to improve the feature screening capability. It achieves detecting the tapping surface of different tapping ages and locating the key points of rubber tapping in the complex rubber forest environment, limited storage and computation capacity, with a view to providing a visual guarantee for intelligent rubber-tapping equipment. The main conclusions are as follows:</p>
<list list-type="simple">
<list-item>
<p>(1) In the ablation experiment, compared with the YOLOv8n-Pose, the YOLOv8n-IRP has been significantly improved in all aspects of accuracy metrics, in which D_P, P_P, D_R, P_R, D_mAP50, and P_mAP50 have been improved by 2.2%, 2.7%, 1.3%, 2.2%, 1.4%, and 2.3%, respectively. The increase in Params and GFlops and the decrease in FPS are inevitable because the AFPN structure performs feature fusion multiple times in adjacent layers to reduce the loss and degradation of low-level and high-level feature information. Considering the actual tapping speed during rubber tapping, 91f/s is sufficient to meet the rubber tapping requirements. Therefore, it is meaningful to significantly improve the detection accuracy of the rubber tree tapping surface and key points while ensuring that the detection speed meets the rubber tapping requirements.</p>
</list-item>
<list-item>
<p>(2) In the comparative experiment of the detection performance of different models, the D_mAP50 and P_mAP50 of YOLOv8n-IRP reach 98.3% and 86.4%, respectively. The visualization results show that for rubber trees of different tapping ages, the confidence of the tapping surface detection is above 96%, and the unobstructed tapping key points can be detected. The overall detection performance is better than that of YOLOv8n-Pose and YOLOv5n-Pose, which meet the requirements of rubber tapping. Although the Faster_RCNN-RTMPose showed the best detection accuracy, it greatly lost model size and computing power, which is not conducive to deployment in mobile rubber tapping equipment, and the detection speed is not enough to meet the requirements of rubber tapping. Therefore, it is further proved that the YOLOv8n-IRP proposed in this paper is more suitable for intelligent rubber tapping.</p>
</list-item>
<list-item>
<p>(3) In the comparative experiment of positioning performance of different models, the average error between the predicted points of the YOLOv8n-IRP and the corresponding truth points in the Euclidean direction was kept within 40 pixels, which was reduced by 12.96 pixels and 15.08 pixels compared with the YOLOv8n-Pose and YOLOv5n-Pose respectively. The visualization results show that for rubber trees of different tapping ages, the predicted points are close to the truth points, with small fluctuations and stable positioning. The overall positioning performance is similar to the Faster_RCNN-RTMPose, better than the YOLOv8n-Pose and YOLOv5n-Pose, and meets the requirements of rubber tapping.</p>
</list-item>
</list>
<p>At present, the method proposed in this paper can accurately detect the tapping surface of natural rubber trees in Danzhou, Hainan. Further research is needed to detect different varieties of rubber trees in other regions, and the positioning accuracy needs to be improved further. In future research, we will collect images of rubber trees of different varieties in different regions, expand the rubber tree data set under different environmental conditions, and study methods to further optimize the network structure and improve the positioning performance. In the entire rubber tapping process, due to the uncertainty of the posture of the rubber trunk, the uncertainty of the attitude of the rubber tree trunk makes it difficult to adjust the end attitude of the robotic arm, so the research on estimating the end attitude of the robotic arm using machine vision is of great significance. Meanwhile, with the integration of different algorithms, the deployment of algorithm models will also bring new challenges, which have higher requirements on the hardware of the rubber-tapping robot, so the research on the lightweight of the model is also of great significance.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>XZ: Conceptualization, Funding acquisition, Methodology, Resources, Writing &#x2013; review &amp; editing. WM: Data curation, Formal analysis, Visualization, Writing &#x2013; original draft. JL: Investigation, Writing &#x2013; review &amp; editing, Supervision. RX: Investigation, Software, Writing &#x2013; review &amp; editing. XC: Software, Visualization, Writing &#x2013; review &amp; editing. YL: Validation, Writing &#x2013; review &amp; editing. ZZ: Methodology, Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by the Hainan Provincial Science and Technology Talent Innovation Project (KJRC2023C04), the National Natural Science Foundation of China (U23A20176), and the National Modern Agricultural Industry Technology System Project (CARS-33-JX2).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altalak</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Uddin</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Alajmi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Rizg</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Smart agriculture applications using deep learning technologies: A survey</article-title>. <source>Appl. Sci.</source> <volume>12</volume>, <fpage>5919</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app12125919</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Arjun</surname> <given-names>R. N.</given-names>
</name>
<name>
<surname>Soumya</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Vishnu</surname> <given-names>R. S.</given-names>
</name>
<name>
<surname>Bhavani</surname> <given-names>R. R.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Semi automatic rubber tree tapping machine</article-title>,&#x201d; in <conf-name>2016 International Conference on Robotics and Automation for Humanitarian Applications (RAHA)</conf-name>, <conf-loc>Amritapuri, India</conf-loc>. <fpage>1</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bello</surname> <given-names>R. W.</given-names>
</name>
<name>
<surname>Oladipo</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Mask YOLOv7-based drone vision system for automated cattle detection and counting</article-title>,&#x201d; in <source>Artificial Intelligence and Applications</source>, <publisher-name>Bon View Publishing Pte. Ltd.</publisher-name>, <publisher-loc>Singapore</publisher-loc>. <fpage>1</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Efficient and lightweight grape and picking point synchronous detection model based on key point detection</article-title>. <source>Comput. Electron. Agric.</source> <volume>217</volume>, <fpage>108612</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.108612</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Weed detection in sesame fields using a YOLO model with an enhanced attention mechanism and feature fusion</article-title>. <source>Comput. Electron. Agric.</source> <volume>202</volume>, <fpage>107412</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.107412</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Tapped area detection and new tapping line location for natural rubber trees based on improved mask region convolutional neural network</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>1038000</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.1038000</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guan</surname> <given-names>Z. B.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y. Q.</given-names>
</name>
<name>
<surname>Chai</surname> <given-names>X. J.</given-names>
</name>
<name>
<surname>Xin</surname> <given-names>C. H. A. I.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J. H.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Visual learning graph convolution for multi-grained orange quality grading</article-title>. <source>J. Integr. Agric.</source> <volume>22</volume>, <fpage>279</fpage>&#x2013;<lpage>291</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jia.2022.09.019</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Las Vegas, NV, USA</conf-loc>. <fpage>770</fpage>&#x2013;<lpage>778</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hou</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Coordinate attention for efficient mobile network design</article-title>,&#x201d; in <conf-name>2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Nashville, TN, USA</conf-loc>. <fpage>13708</fpage>&#x2013;<lpage>13717</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Kukreja</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Image-based wheat mosaic virus detection with Mask-RCNN model</article-title>,&#x201d; in <conf-name>2022 International Conference on Decision Aid Sciences and Applications (DASA)</conf-name>, <conf-loc>Chiangrai, Thailand</conf-loc>. <fpage>178</fpage>&#x2013;<lpage>182</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>W. S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Immature green citrus fruit detection and counting based on fast normalized cross correlation (FNCC) using natural outdoor colour images</article-title>. <source>Precis. Agric.</source> <volume>17</volume>, <fpage>678</fpage>&#x2013;<lpage>697</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11119-016-9443-z</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Selective kernel networks</article-title>,&#x201d; in <conf-name>2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Long Beach, CA, USA</conf-loc>. <fpage>510</fpage>&#x2013;<lpage>519</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>He</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Scconv: spatial and channel reconstruction convolution for feature redundancy</article-title>,&#x201d; in <conf-name>2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Vancouver, BC, Canada</conf-loc>. <fpage>6153</fpage>&#x2013;<lpage>6162</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Fruit detection in natural environment using partial shape matching and probabilistic Hough transform</article-title>. <source>Precis. Agric.</source> <volume>21</volume>, <fpage>160</fpage>&#x2013;<lpage>177</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11119-019-09662-w</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>The vision-based target recognition, localization, and control for harvesting robots: A review</article-title>. <source>Int. J. Precis. Eng. Manufacturing</source> <volume>25</volume>, <fpage>409</fpage>&#x2013;<lpage>428</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12541-023-00911-7</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Luo</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Learning optical flow with kernel patch attention</article-title>,&#x201d; in <conf-name>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>New Orleans, LA, USA</conf-loc>. <fpage>8896</fpage>&#x2013;<lpage>8905</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mokayed</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Quan</surname> <given-names>T. Z.</given-names>
</name>
<name>
<surname>Alkhaled</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Sivakumar</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Real-time human detection and counting system using deep learning computer vision techniques</article-title>,&#x201d; in&#xa0;<source>Artificial Intelligence and Applications</source>, <publisher-name>Bon View Publishing Pte. Ltd.</publisher-name>, <publisher-loc>Singapore</publisher-loc>. <fpage>221</fpage>&#x2013;<lpage>229</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ortatas</surname> <given-names>F. N.</given-names>
</name>
<name>
<surname>Ozkaya</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Sahin</surname> <given-names>M. E.</given-names>
</name>
<name>
<surname>Ulutas</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Sugar beet farming goes high-tech: a method for automated weed detection using machine learning and deep learning in precision agriculture</article-title>. <source>Neural Computing Appl.</source> <volume>36</volume>, <fpage>4603</fpage>&#x2013;<lpage>4622</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00521-023-09320-3</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J. Y.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A simple and light-weight attention module for convolutional neural networks</article-title>. <source>Int. J. Comput. Vision</source> <volume>128</volume>, <fpage>783</fpage>&#x2013;<lpage>798</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11263-019-01283-0</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rehman</surname> <given-names>T. U.</given-names>
</name>
<name>
<surname>Mahmud</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>Y. K.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Current and future applications of statistical machine learning algorithms for agricultural machine vision systems</article-title>. <source>Comput. Electron. Agric.</source> <volume>156</volume>, <fpage>585</fpage>&#x2013;<lpage>605</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2018.12.006</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>C. Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J. S.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>J. Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y. A. N. G.</given-names>
</name>
<name>
<surname>Hang</surname> <given-names>Z. H. O. U.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Detection of maize tassels for UAV remote sensing image with an improved YOLOX model</article-title>. <source>J. Integr. Agric.</source> <volume>22</volume>, <fpage>1671</fpage>&#x2013;<lpage>1683</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jia.2022.09.021</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Soumya</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Vishnu</surname> <given-names>R. S.</given-names>
</name>
<name>
<surname>Arjun</surname> <given-names>R. N.</given-names>
</name>
<name>
<surname>Bhavani</surname> <given-names>R. R.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Design and testing of a semi-automatic rubber tree tapping machine</article-title>,&#x201d; in <conf-name>2016 IEEE Region 10 Humanitarian Technology Conference (R10-HTC)</conf-name>, <conf-loc>Agra, India</conf-loc>. <fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An improved YOLOv5-based tapping trajectory detection method for natural rubber trees</article-title>. <source>Agriculture</source> <volume>12</volume>, <fpage>1309</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture12091309</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Advances in genome sequencing and natural rubber biosynthesis in rubber-producing plants</article-title>. <source>Curr. Issues Mol. Biol.</source> <volume>45</volume>, <fpage>9342</fpage>&#x2013;<lpage>9353</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cimb45120585</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>W. S.</given-names>
</name>
<name>
<surname>Gan</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Recognising blueberry fruit of different maturity using histogram oriented gradients and colour features in outdoor scenes</article-title>. <source>Biosyst. Eng.</source> <volume>176</volume>, <fpage>59</fpage>&#x2013;<lpage>72</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2018.08.011</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Improved multi-scale inverse bottleneck residual network based on triplet parallel attention for apple leaf disease identification</article-title>. <source>J. Integr. Agric.</source> <volume>23</volume>, <fpage>901</fpage>&#x2013;<lpage>922</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jia.2023.06.023</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thakur</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Venu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gurusamy</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An extensive review on agricultural robots with a focus on their perception systems</article-title>. <source>Comput. Electron. Agric.</source> <volume>212</volume>, <fpage>108146</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108146</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>b). <article-title>Peanut defect identification based on multispectral image and deep learning</article-title>. <source>Agronomy</source> <volume>13</volume>, <fpage>1158</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agronomy13041158</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>He</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>c). <article-title>An energy-efficient classification system for peach ripeness using YOLOv4 and flexible piezoelectric sensor</article-title>. <source>Comput. Electron. Agric.</source> <volume>210</volume>, <fpage>107909</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.107909</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>a). <article-title>A performance analysis of a litchi picking robot system for actively removing obstructions, using an artificial intelligence algorithm</article-title>. <source>Agronomy</source> <volume>13</volume>, <fpage>2795</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agronomy13112795</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J. Y.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Cbam: Convolutional block attention module</article-title>,&#x201d; in <conf-name>Proceedings of the European conference on computer vision (ECCV)</conf-name>, <conf-loc>Cham</conf-loc>. <fpage>3</fpage>&#x2013;<lpage>19</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Afpn: Asymptotic feature pyramid network for object detection</article-title>. <source>arXiv preprint arXiv:2306.15988</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2306.15988</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>a). <article-title>Rfaconv: Innovating spatital attention and standard convolutional operation</article-title>. <source>arXiv preprint arXiv:2304.03198</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2304.03198</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xuan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Xue</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>b). <article-title>LSR-YOLO: A high-precision, lightweight model for sheep face recognition on the mobile end</article-title>. <source>Animals</source> <volume>13</volume>, <fpage>1824</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ani13111824</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Design, development, and field evaluation of a rubber tapping robot</article-title>. <source>J. Field. Robot</source> <volume>39</volume>, <fpage>28</fpage>&#x2013;<lpage>54</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/rob.22036</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Design, development, and field evaluation of a rubber tapping robot</article-title>. <source>J. Field Robotics</source> <volume>39</volume>, <fpage>28</fpage>&#x2013;<lpage>54</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/rob.22036</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>