<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2023.1230517</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Application of improved YOLOv7-based sugarcane stem node recognition algorithm in complex environments</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wen</surname>
<given-names>Chunming</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guo</surname>
<given-names>Huanyu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2316700"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Jianheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2384204"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hou</surname>
<given-names>Bingxu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2383991"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Youzong</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Kaihua</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nong</surname>
<given-names>Hongliang</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Long</surname>
<given-names>Xiaozhu</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lu</surname>
<given-names>Yuchun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Electronic Information, Guangxi Minzu University</institution>, <addr-line>Nanning</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Guangxi Key Laboratory of Intelligent Unmanned System and Intelligent Equipment</institution>, <addr-line>Nanning, Guangxi</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Guangxi Key Laboratory of Hybrid Computation and IC Design Analysis</institution>, <addr-line>Nanning, Guangxi</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>State Key Laboratory for Conservation and Utilization of Subtropical Agro-bioresources</institution>, <addr-line>Nanning, Guangxi</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Technology Development Center, Guangxi Agricultural Machinery Research Institute</institution>, <addr-line>Nanning</addr-line>, <country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Technical Research and Development, Nanning Titanium Silver Technology Co.</institution>, <addr-line>Nanning</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Chuanlei Zhang, Tianjin University of Science and Technology, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Chunlei Xia, Chinese Academy of Sciences (CAS), China; Naveen Kumar Mahanti, Dr.Y.S.R. Horticultural University, India; Ning Yang, Jiangsu University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Chunming Wen, <email xlink:href="mailto:wenchunming@gxmzu.edu.cn">wenchunming@gxmzu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1230517</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>05</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>07</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Wen, Guo, Li, Hou, Huang, Li, Nong, Long and Lu</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Wen, Guo, Li, Hou, Huang, Li, Nong, Long and Lu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Sugarcane stem node detection is one of the key functions of a small intelligent sugarcane harvesting robot, but the accuracy of sugarcane stem node detection is severely degraded in complex field environments when the sugarcane is in the shadow of confusing backgrounds and other objects.</p>
</sec>
<sec>
<title>Methods</title>
<p>To address the problem of low accuracy of sugarcane arise node detection in complex environments, this paper proposes an improved sugarcane stem node detection model based on YOLOv7. First, the SimAM (A Simple Parameter-Free Attention Module for Convolutional Neural Networks) attention mechanism is added to solve the problem of feature loss due to the loss of image global context information in the convolution process, which improves the detection accuracy of the model in the case of image blurring; Second, the Deformable convolution Network is used to replace some of the traditional convolution layers in the original YOLOv7. Finally, a new bounding box regression loss function WIoU Loss is introduced to solve the problem of unbalanced sample quality, improve the model robustness and generalization ability, and accelerate the convergence speed of the network.</p>
</sec>
<sec>
<title>Results</title>
<p>The experimental results show that the mAP of the improved algorithm model is 94.53% and the F1 value is 92.41, which are 3.43% and 2.21 respectively compared with the YOLOv7 model, and compared with the mAP of the SOTA method which is 94.1%, an improvement of 0.43% is achieved, which effectively improves the detection performance of the target detection model.</p>
</sec>
<sec>
<title>Discussion</title>
<p>This study provides a theoretical basis and technical support for the development of a small intelligent sugarcane harvesting robot, and may also provide a reference for the detection of other types of crops in similar environments.</p>
</sec>
</abstract>
<kwd-group>
<kwd>sugarcane stem node detection</kwd>
<kwd>SimAM</kwd>
<kwd>deformable convolution</kwd>
<kwd>WIoU</kwd>
<kwd>YOLOv7</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="3"/>
<equation-count count="14"/>
<ref-count count="25"/>
<page-count count="14"/>
<word-count count="6167"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Sugarcane is the main raw material for sugar production. Although China&#x2019;s sugarcane cultivation area is large, it is mostly planted in hilly areas, which is not conducive to the work of existing large and medium sized sugarcane harvesters (<xref ref-type="bibr" rid="B21">Zeng et&#xa0;al., 2012</xref>). Therefore, the research of miniaturized and intelligent sugarcane harvesters is a development trend, and the recognition of sugarcane stem nodes to judge the position of sugarcane nodes cutting is the first step to realizing intelligent sugarcane harvesting operation.</p>
<p>In the study of recognition of sugarcane stem nodes, <xref ref-type="bibr" rid="B11">Moshashai et&#xa0;al. (2008)</xref> first investigated the recognition method of sugarcane stem nodes by comparing the diameters of different parts of sugarcane. <xref ref-type="bibr" rid="B9">Lu et&#xa0;al. (2010)</xref> proposed a support vector machine-based feature extraction and recognition method for sugarcane stem nodes, and the recognition rate reached 94.118%. <xref ref-type="bibr" rid="B7">Huang et&#xa0;al. (2013)</xref> searched the edges of grayscale images by soble operator and achieved 100% recognition rate in detecting and locating sugarcane nodes using random transform. <xref ref-type="bibr" rid="B10">Meng et&#xa0;al. (2019)</xref> proposed a sugarcane stem node recognition algorithm based on multi-threshold and multi-scale wavelet transform, applied to stem node recognition of leaf stripped sugarcane with 100% recognition rate. <xref ref-type="bibr" rid="B24">Zhou et&#xa0;al. (2020)</xref> proposed a sugarcane stem node recognition based on Sobel edge detection based sugarcane stem node recognition method, the recognition rate of 93% can meet the working requirements of a sugarcane seed-cutting machine. <xref ref-type="bibr" rid="B3">Chen et&#xa0;al. (2021a)</xref> proposed a sugarcane node recognition algorithm based on the minimum point local pixel sum of vertical projection function and analyzed the recognition of single and double nodes, where the recognition rate of a single node is 100% and the recognition rate of double nodes is 98.5%. The above methods mainly rely on traditional image processing methods, which need to work in simple environments and cannot meet the requirements of real-time detection in complex backgrounds.</p>
<p>In recent years, with the development of deep learning technology and the continuous open source of classical target detection algorithms such as Faster-RCNN and YOLO series, the algorithms of deep learning have been widely used in the field of agriculture. As a single-phase detection method, the YOLO algorithm has the characteristics of fast speed and high efficiency compared to the two-phase detection method, which is widely used in target detection in real scenes. <xref ref-type="bibr" rid="B22">Zhang et&#xa0;al. (2022)</xref> proposed an improved lightweight network based on YOLOv5s to achieve all-weather detection of dragon fruit in complex orchard environments, introduced a ghost module in yolov5 to realize the lightweight of the model, added a coordinate attention mechanism so that the model can accurately locate and identify dense dragon fruit, and adopted the SIoU loss function to improve the convergence speed, and the results show that the average precision (mAP) of the model is 97.4%. <xref ref-type="bibr" rid="B20">Zang et&#xa0;al. (2022)</xref> proposed an improved attention mechanism based on YOLOv5s for detecting the number of small-scale wheat spikes and better solving the problem of occlusion and cross-overlapping of wheat spikes. The university&#x2019;s channel attention module (ECA) is introduced in the C3 module of the YOLOv5 backbone structure. The results show that the improved YOLOv5s model achieves an accuracy of 71.61% in the wheat spike counting task, which is 4.95% higher than the original model, and the method improves the applicability in complex field environments. <xref ref-type="bibr" rid="B15">Wang et&#xa0;al. (2022a)</xref> proposed an improved YOLOv4 model for the accurate detection of pear blossoms in natural environments, which consists of the SENet (squeeze - and - excitation Networks) module-embedded ShuffleNetv2 replaces the original backbone network of the YOLOv4 model and constitutes the backbone network of the YOLO-PEFL model. The experimental results show that the YOLO-PEFL model has an average accuracy of 96.71% and can accurately detect pear blossoms in the natural environment. <xref ref-type="bibr" rid="B8">Li et&#xa0;al. (2019)</xref> established an intelligent recognition convolutional neural network model by improving the YOLOv3 network, and the recognition accuracy of stem nodes was 96.89%, however, sugarcane samples were preprocessed by manually removing leaves in a preprocessed monochromatic background environment. To promote sugarcane precut seeds good seeds and good method planting technology, <xref ref-type="bibr" rid="B14">Wang et&#xa0;al. (2022b)</xref> proposed an algorithm to improve YOLOv4-Tiny to achieve accurate and fast identification and cutting of sugarcane stem nodes, and the detection accuracy of the improved algorithm was 97.07%.<xref ref-type="bibr" rid="B25">Zhu et&#xa0;al. (2022)</xref> proposed a new method of binocular localization based on improved YOLOv4 for the difficult spatial localization of sugarcane nodes using robots under agricultural conditions, and lightened YOLOv4 for porting to embedded chips by network slimming techniques. The results showed that the improved YOLOv4 algorithm reduced the model size, parameters, and FlOPs by about 89.1%, which greatly reduced the complexity of the model. The complexity of the model is greatly reduced, but the accuracy is also slightly reduced. <xref ref-type="bibr" rid="B2">Chen et&#xa0;al. (2021b)</xref> proposed a deep learning-based target detection algorithm for the problem of low accuracy of sugarcane stem node recognition in the natural environment and improved the robustness and generalization ability of the algorithm by data set expansion method, and the results showed that the average accuracy was 95.17%. Although good accuracy is obtained, the use of the dataset expansion method is likely to lead to data over fitting.</p>
<p>Deep learning-based target detection methods have already achieved good results in sugarcane stem node recognition in simple backgrounds, but in the field, there is the problem of difficulty in recognizing and accurately locating sugarcane stem nodes in complex environments constituted by light, shading, and other characteristics of the crop such as dense sugarcane and different maturity levels. Therefore, a target detection model based on improved YOLOv7 is proposed in this paper, and the main contributions are as follows:</p>
<list list-type="order">
<list-item>
<p>Establishing sugarcane datasets in complex environments.</p>
</list-item>
<list-item>
<p>In this paper, an improved target detection network model for complex environments based on YOLOv7 is proposed. The Deformable Convolution Network (DCN) is introduced to replace part of the convolutional layer of the feature extraction network in the original YOLOv7 so that the feature extraction network can adaptively extract the positional features such as occlusion and overlap that leads to the lack of information of sugarcane stems and nodes and the SimAM is added to the ELAN module in the backbone network and the concatenation layer in the feature fusion module. An attention mechanism is added to the ELAN module in the backbone network and the Concat connection layer in the feature fusion module to enhance the extraction ability of the model for small and dense sugarcane stem node features without increasing the complexity of the model. The WIoU loss function is used instead of the original loss function to solve the sample quality imbalance problem, which improves the convergence speed of the model during training.</p>
</list-item>
<list-item>
<p>The superiority of our model in the task of target detection in complex environments is verified from different perspectives through comparative and ablation experiments, and the experiments provide ideas and rationale for small intelligent sugarcane harvesters to recognize sugarcane stem nodes.</p>
</list-item>
</list>
</sec>
<sec id="s2">
<label>2</label>
<title>YOLOV7 network model and improvements</title>
<sec id="s2_1">
<label>2.1</label>
<title>YOLOv7 model</title>
<p>The YOLOv7 (<xref ref-type="bibr" rid="B13">Wang et&#xa0;al., 2023</xref>) target detection algorithm, introduced by the original YOLOv4 (<xref ref-type="bibr" rid="B1">Bochkovskiy et&#xa0;al., 2020</xref>) research team in July 2022, is a novel and excellent detector that uses instead of efficient aggregation network, the ELAN module that appears in the network structure, to effectively enhance the network learning capability compared to the previous YOLO series. The YOLOv7 network structure mainly consists of the Input layer, Backbone layer, Neck layer and Head layer, Backbone layer, Neck layer, and Head layer. The input layer is the input layer, which scales the input image to a fixed size to meet the input size requirement of Backbone. The backbone layer is the feature extraction layer, and based on YOLOv5, ELAN structure, and MP structure are introduced. Among them, the ELAN structure consists of different convolutional blocks stacked without changing the width-height of the input feature layers and enhances the interaction between each feature layer through expansion, random combination, and splicing to improve the learning ability of the model. The MP structure consists of convolutional blocks of 3x3 size and a Maxpool dual path, which compresses the width-height of the input feature layers to enhance the feature fusion ability of the network. The neck feature fusion network (Neck layer) includes CBS, SPPCSPC, MP, and ELAN, which follow the traditional PAFPN structure to extract three feature layers located in the middle, lower middle, and bottom layers of the backbone part, respectively, to achieve full fusion of multi-scale features. the SPPCSPC structure achieves a full fusion of multi-scale features by introducing a convolutional spatial pyramid (CSP) in the spatial pyramid pool (SPP) structure) structure to improve the perceptual field of the network, while multiple pooling operations are added in parallel in a string of convolutions, using residual edges for optimization and feature extraction. The Head layer uses the anchor mechanism to output feature maps at three scales, large and small, and uses the reparameterized structure RepConv to articulate the regular convolution to adjust the number of channels and prediction, and then the final prediction results are obtained through the processing of CIou loss function and nonlinear maxima suppression (NMS).</p>
<p>Although the YOLOv7 algorithm performs well in common task scenarios (e.g. pedestrian and vehicle detection), there are still many problems in applying it directly to sugarcane stem node recognition in complex environments: for example, in the actual sugarcane environment, dense clusters of sugarcane, small stem node size, and a large number, and a certain degree of occlusion or overlap will lead to serious cases of missed and false detection. To address the above problems, this paper improves the YOLOv7 algorithm in terms of attention mechanism, convolution layer, and loss function to improve the recognition effect in complex environments. The YOLOv7 network structure is shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>YOLOv7 network structure diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g001.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>SimAM attention mechanism</title>
<p>The attention mechanism refers to the model ignoring irrelevant information and focusing on important information by assigning different weights to the input parts of the network, which can effectively improve the feature extraction ability of the model in complex backgrounds. pyramid feature extraction is used as the backbone network in YOLOv7, and in response to the characteristics of dense sugarcane, small target, large number, and easy to obscure and overlap in the natural environment, this paper adds an attention mechanism to the backbone In this paper, we add an attention mechanism to the backbone network to improve the feature extraction capability and enhance the feature representation capability. The traditional attention mechanisms SE (<xref ref-type="bibr" rid="B6">Hu et&#xa0;al., 2018</xref>) (Squeeze-and-Excitation), (<xref ref-type="bibr" rid="B18">Woo et&#xa0;al., 2018</xref>) CBAM (Convolutional Block Attention Module), ECA (<xref ref-type="bibr" rid="B16">Wang et&#xa0;al., 2020</xref>) (Efficient Channel Attention Module), CA (<xref ref-type="bibr" rid="B5">Hou et&#xa0;al., 2021</xref>) (Coordinate Attention), all assign attention weights along channels or spatial locations and require additional sub-network structures and model parameters, which on the one hand cannot generate real 3D weights based on channels or spaces, and on the other hand, the additional sub-network structures inevitably lead to an increase in network complexity. In order to solve the shortage of traditional attention mechanisms, this paper proposes to use SimAM (<xref ref-type="bibr" rid="B19">Yang et&#xa0;al., 2021</xref>) non-parametric attention mechanism, whose attention weights are assigned as shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>SimAM attention mechanism.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g002.tif"/>
</fig>
<p>The SimAM attention mechanism generates truly effective 3D weights directly by designing an energy function, without adding additional sub-networks or additional model parameters. The design of the energy function is inspired by neuroscience theory and aims to measure the linear differentiability between neurons to find the important neurons, and for each neuron of the input, the minimum energy function is defined as follows.</p>
<disp-formula>
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi>&#x3c3;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
<mml:msup>
<mml:mover accent="true">
<mml:mi>&#x3c3;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>t</italic> denotes the target neuron of the input feature in the current channel, where <italic>&#xb5;</italic>,<italic>&#x3c3;</italic>
<sup>2</sup>, is the mean and variance of all neurons in the channel to avoid repeated calculations and reduce the computational cost, and <italic>&#x3bb;</italic> is the weight constant. To better realize attention, the SimAM module needs to assess the importance of each neuron. In neuroscience, neurons with rich information usually exhibit different firing patterns than surrounding neurons. In addition, activated neurons tend to inhibit surrounding neurons, spatial inhibition, and neurons with spatial inhibition should be given higher importance (<xref ref-type="bibr" rid="B17">Webb et&#xa0;al., 2005</xref>). Equation (1) reveals a phenomenon: the lower the energy t of a neuron, the more it differs from surrounding neurons, and therefore the more significant its contribution to visual processing. Thus, the importance of each neuron can be evaluated in terms of <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. SimAM attention mechanism is added to the ELAN module of the backbone network as well as the Concat connection layer of the neck network, SimAM adjusts the distribution of the attention of the feature map by evaluating the importance of each neuron, so that the channel and the spatial attention work in synergy, the importance of the neurons of the sugarcane stem node will be calculated through the training model, and the information of the secondary disturbances other than the sugarcane stem node will be suppressed, so that the information of the secondary disturbances other than the sugarcane stem node will be suppressed, thus weaken the influence of complex environmental factors on sugarcane stem node recognition. At the same time, SimAM can adaptively adjust the weights of feature mapping and pay more attention to the local area of the target. This can improve the target localization accuracy, reduce the localization error and enhance the feature extraction ability of the backbone network.</p>
<p>The output equation of the attention module is:</p>
<disp-formula>
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2609;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The final output is obtained by adding the Sigmoid function to suppress the outliers of the attention weights and performing the dot product operation with the corresponding elements of the input feature matrix.</p>
<p>Aiming at the recognition of sugarcane stem nodes in complex environments and the lack of an attention mechanism in the YOLOv7 network, this paper effectively extracts finer-grained feature information by adding the SimAM attention mechanism to the last layer of the 1&#xd7;1 CBS of the ELAN module in the Backbone module of YOLOv7 and by incorporating the Concat in the Neck into the SimAM attention mechanism.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Deformable convolution</title>
<p>In the original YOLOv7 model, ordinary convolutional blocks are used in the feature extraction network Backbone, which mainly consists of a traditional convolutional layer, BN layer, and activation function, and due to the fixed size of the convolutional kernel of the traditional convolutional layer, it has poor robustness to unknown geometric transformations and poor generalization ability. When performing feature extraction, it is difficult for the fixed-size convolution kernel to extract the boundary information of the object accurately because the size and contour of different objects are generally different, which affects the ability of the network to extract the features of the object. The traditional convolutional calculation method is as follows:</p>
<disp-formula>
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>W</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>R</italic> denotes the size of the partial feature map corresponding to the convolution kernel, <italic>W</italic> (<italic>p<sub>n</sub>
</italic>) denotes the weight corresponding to the nth sampled point in the sampled region, and <italic>X</italic> (<italic>p</italic>
<sub>0</sub> + <italic>p<sub>n</sub>
</italic>) denotes the pixel value size of the nth sampled point. To solve the above problem, a deformable convolutional network (<xref ref-type="bibr" rid="B4">Dai et&#xa0;al., 2017</xref>) is introduced in this paper, as shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>:</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Deformable convolutional network diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g003.tif"/>
</fig>
<p>The network is able to adaptively adjust the size and shape of the convolution kernel for objects of different shapes, which has better robustness and generalization ability compared with traditional convolutional networks, thus enhancing the underlying network&#x2019;s ability to extract object features. In the variable convolution operator, adding a learnable offset parameter offset to each element in the convolution kernel can make the originally fixed convolution kernel have the ability to adapt to the object shape, and the computational equation is as follows:</p>
<disp-formula>
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>W</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x394;</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the variable convolution calculation equation (3), &#x394;<italic>p<sub>n</sub>
</italic>denotes the offset. Considering the irregularity of the sampling position of the variable convolution, so that the offset is generally fractional, equation (4) is implemented with bilinear interpolation as follows:</p>
<disp-formula>
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>q</mml:mi>
</mml:munder>
<mml:mi>G</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>q</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>p</italic> denotes an arbitrary position in the region corresponding to the deformable convolution, <italic>q</italic> is the pixel value corresponding to a sampling point in the feature map <italic>X</italic>, and <italic>G</italic>(<italic>q,p</italic>) denotes a two-dimensional bilinear interpolation kernel. In the sugarcane stem node detection task, sugarcane may have complex deformations, such as cane twisting, deformation, or attitude changes, which have an impact on the accurate identification of sugarcane stem nodes. Traditional fixed convolution kernels are difficult to capture these local deformations. Deformable convolution adjusts the sampling position of the convolution kernel by introducing offsets, which allows the convolution operation to better adapt to the target&#x2019;s deformations and enhances the model&#x2019;s ability to recognize complex shapes. Deformable convolution utilizes additional convolutional layers to learn the corresponding offsets, and superimposes the obtained offsets on the corresponding pixels in the input feature maps, allowing the convolutional kernel to diverge the sampling in the input feature maps, so that the network can focus on the target function. The embedded deformable convolutional layer can adaptively adjust the sensory field size and position during the convolution process, so that the sampling position around each position during pooling is adaptive and better adapts to the shape and size of the sugarcane stem nodes, thus improving the detection accuracy.</p>
<p>Since deformable convolution adds one more parameter compared to conventional convolution, the 3 &#xd7; 3 convolution kernel in the ElAN module of the backbone network is replaced with deformable convolution (DCN) in the improved algorithm of this paper, and DCN, BN, and SiLU form the DBS module. And the last layer of 1 &#xd7; 1 CBS in ELAN is replaced with the SimAM attention mechanism, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, and the replaced structure is represented by DS-ELAN, which enhances the network with a smaller increase in the computational capability of feature extraction and complex background target detection.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>DS-ELAN structure diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g004.tif"/>
</fig>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Loss function improvement</title>
<p>The YOLOv7 network uses CIoU (<xref ref-type="bibr" rid="B23">Zheng et&#xa0;al., 2021</xref>) Loss as the Bounding Box loss function of the model. This function is mainly used for the regression of the prediction box so that the prediction box of the object is closer to the position of the object labeled bounding box. In the model training, CIoU Loss calculates the distance between the prediction box and the center of the real bounding box, the overlap area, and the aspect ratio of the two boxes for the regression of the bounding box, but it does not take into account the balance of the quality of the training samples, which will lead to the slow convergence and low efficiency of the network, and may result in a worse model due to the random matching of the prediction boxes during training. CIoU is calculated as follows:</p>
<disp-formula>
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>4</mml:mn>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>arctan</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>arctan</mml:mi>
<mml:mfrac>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where: <italic>b</italic> denotes the centroid of the prediction frame; <italic>b</italic>
<sub>gt</sub> denotes the centroid of the true frame; <italic>&#x3c1;</italic> represents the Euclidean distance between the two centroids is calculated, and <italic>c</italic> denotes the diagonal length of the minimum enclosing frame covering the prediction frame and the true frame;<italic>&#x3b1;</italic> is the balance parameter; <italic>v</italic> is used to measure whether the aspect ratio is consistent. It can be seen from Equation (7) that when the aspect ratio of the prediction frame and the true value are equal and v takes 0, the penalty term for the aspect ratio in the loss function in CIoU degenerates to 0, resulting in a penalty failure and the final prediction frame cannot fit the true frame. When the prediction frame fits the target frame well, a good loss function should be able to attenuate the penalty of geometric factors.</p>
<p>Because the training data inevitably contains low-quality examples, geometric measures such as distance and aspect ratio can exacerbate the penalty on low-quality examples thus degrading the generalization performance of the model.Wise Iou (<xref ref-type="bibr" rid="B12">Tong et&#xa0;al., 2023</xref>) can well improve the sample quality imbalance problem and increase the accuracy of the target detection algorithm. The WIoU loss function is formulated as follows:</p>
<disp-formula>
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>v</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>*</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>R<sub>WIoU</sub>
</italic>range is [1,e), which significantly amplifies the <italic>L<sub>IoU</sub>
</italic>of the common quality anchor box.<italic>L<sub>IoU</sub>
</italic>range is [0,1]. will significantly reduce the <italic>L<sub>IoU</sub>
</italic>of the high-quality anchor frame and the distance between its center of attention when the anchor frame overlaps well with the target frame.</p>
<p>where <italic>x y</italic> is the coordinate of the center point of the prediction frame, <italic>x</italic>
<sub>gt,</sub>
<italic>y</italic>
<sub>gt</sub> is the coordinate of the center point of the real frame, and <italic>Wg</italic>, <italic>Hg</italic> denote the width and height of the minimum enclosing frame. In order to prevent the generation of gradients that hinder convergence, <italic>Wg</italic> and <italic>Hg</italic> are separated from the computational graph (the superscript * indicates this operation), which effectively eliminates the factors that hinder convergence, so no new metric, such as aspect ratio, is introduced. Since L<italic>
<sub>IoU</sub>
</italic>is dynamic, the quality classification criteria of the anchor boxes are also dynamic, which allows WIoU to make a gradient gain allocation strategy that best fits the current situation at each moment.</p>
<p>This strategy reduces the competitiveness of high-quality anchor boxes and also reduces the deleterious gradients generated by low-quality samples. In the sugarcane stem node detection task, some samples may be challenging due to the diversity and complexity of sugarcane images, such as blurred, occluded, or small size of sugarcane images. For these low-quality examples, they may generate noisy or unreliable gradient signals that interfere with the model training process. By introducing category weights through multiple, the WIoU loss function can reduce the weights of low-quality samples relatively, which reduces the impact of these samples on the model parameter updates. Therefore, by reducing the competitiveness of high-quality anchor boxes and reducing the harmful gradient of low-quality samples, the WIoU loss function is able to focus more on the optimization of average-quality anchor boxes and improve the performance of the target detector in general. This strategy helps to make the model more focused on the detection accuracy of important target categories and enhances its ability to handle medium-quality samples, thus improving the effectiveness and performance of overall target detection.</p>
<p>Therefore, this paper uses the WIoU loss function to replace the CIoU in the original network to solve the problem that the regression boxes cannot be matched accurately due to the unbalanced sample quality.</p>
<p>In summary, the improved algorithm adds the SimAM attention mechanism to the feature fusion part of the original YOLOv7 backbone network to enhance the network&#x2019;s ability to extract target feature information in complex scenes. By using deformable convolution to replace some ordinary convolution layers, the convolution kernel is deformed by more pairs of convolution kernels, which in turn enhances the perceptual field and feature extraction ability of the convolutional network, so that it can better adapt to the shape and position changes of target objects, thus improving the accuracy of object recognition and localization. The CIoU is replaced by WIoU to solve the sample mass balance problem and to make the prediction frame fit the real frame better to improve the detection accuracy. The structure diagram of the improved YOLOv7 network is shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Improved YOLOv7 network structure diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Experiment and analysis</title>
<sec id="s3_1">
<label>3.1</label>
<title>Experimental data set</title>
<p>The sugarcane images used in this paper were taken from Su Village, Tao Wei Town, Hengxian County, Guangxi, China, which is a sugarcane plantation. The images were captured by a Xiaomi 10pro digital camera with a resolution of 1080 &#xd7; 1440 pixels, and the shooting time periods were morning, noon, and afternoon, and a total of 2144 different sugarcane images were obtained by constantly changing the distance and shooting angle. They contain images under uneven conditions of natural scenes such as leaf occlusion, overlapping occlusion, visual similarity to the background image, dense target, backlight, front light, and side light, and are saved in JPG format. The following <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> shows some images taken under different conditions, which were randomly divided into 80% as the training set, 10% as the validation set, and 10% as the test set, forming 1715, 214, and 214 images for model training and testing, respectively. The datasets were then annotated, and the sugarcane stem node bounding boxes in each image were drawn manually using the Roboflow annotation platform, and the annotation files were saved in YOLO text format, which requires the target class, coordinates, height, and width. The process of annotating the sugarcane stem nodes is shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Partial data set display: <bold>(A)</bold> Sunbeam; <bold>(B)</bold> Back lighting; <bold>(C)</bold> Light blocking; <bold>(D)</bold> Leaf wrap; <bold>(E)</bold> Foliage shade; <bold>(F)</bold> Weed shading.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g006.tif"/>
</fig>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Marking process.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g007.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Model evaluation metrics</title>
<p>This paper adopted evaluation metrics including precision (<italic>P</italic>), recall (<italic>R</italic>), mean average precision (<italic>mAP</italic>), and F1 score.</p>
<p>
<italic>P</italic> and <italic>R</italic> refer to the precision and recall of the detection model, respectively. Precision represents the proportion of true positive samples in the samples predicted as positive by the classifier. The recall represents the proportion of true positive samples that are correctly predicted as positive by the classifier among all true positive samples. The formula for calculating precision and recall is:</p>
<disp-formula>
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The <italic>F</italic>1 score considers both precision and recall, and it can reflect the stability of a model. A higher <italic>F</italic>1 score indicates a more stable model. The formula for calculating the <italic>F</italic>1 score is:</p>
<disp-formula>
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>mAP</italic> is the average precision of each class and the average value of <italic>AP</italic>, its calculation formula is:</p>
<disp-formula>
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>C</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Experimental environment and parameter settings</title>
<p>The experimental environment is a 64-bit Ubuntu 22.04 system with an Intel(R) Xeon(R) Platinum 8157 CPU @ 2.30GHz and an NVIDIA GeForce RTX3090 graphics card with 24GB video memory. The study is based on the PyTorch deep learning framework, and the development environment is PyTorch 1.11.0, Cuda 11.3, and Python interpreter version 3.9. The parameters of the experiments conducted in this experiment are shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>:</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Experimental parameters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Parameters</th>
<th valign="top" align="center">Values</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Learn rate</td>
<td valign="top" align="left">0.01</td>
</tr>
<tr>
<td valign="top" align="left">Epochs</td>
<td valign="top" align="left">100</td>
</tr>
<tr>
<td valign="top" align="left">Batch-size</td>
<td valign="top" align="left">16</td>
</tr>
<tr>
<td valign="top" align="left">Wokers</td>
<td valign="top" align="left">4</td>
</tr>
<tr>
<td valign="top" align="left">Img size</td>
<td valign="top" align="left">640&#xd7;640</td>
</tr>
<tr>
<td valign="top" align="left">Nms</td>
<td valign="top" align="left">0.3</td>
</tr>
<tr>
<td valign="top" align="left">Conf thres</td>
<td valign="top" align="left">0.25</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Ablation experiments</title>
<p>To validate the effectiveness of the improvement points in this paper, five sets of ablation experiments were conducted on the sugarcane dataset using the original YOLOv7 network as a baseline and keeping the environment and parameters uniform. The validation criteria include mAP values and F1 values. The experimental results are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, where bold font indicates the optimal results in each column and &#x221a; indicates the use of the corresponding method.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Ablation experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Number</th>
<th valign="top" align="center">SimAM</th>
<th valign="top" align="center">DCN</th>
<th valign="top" align="center">WIoU</th>
<th valign="top" align="center">mAP(%)</th>
<th valign="top" align="center">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">No.1</td>
<td valign="middle" align="left"/>
<td valign="bottom" align="left"/>
<td valign="bottom" align="left"/>
<td valign="top" align="left">91.10</td>
<td valign="top" align="left">90.20</td>
</tr>
<tr>
<td valign="top" align="left">No.2</td>
<td valign="middle" align="left">&#x221a;</td>
<td valign="bottom" align="left"/>
<td valign="bottom" align="left"/>
<td valign="top" align="left">92.50</td>
<td valign="top" align="left">91.13</td>
</tr>
<tr>
<td valign="top" align="left">No.3</td>
<td valign="middle" align="left"/>
<td valign="bottom" align="left">&#x221a;</td>
<td valign="bottom" align="left"/>
<td valign="top" align="left">92.30</td>
<td valign="top" align="left">91.08</td>
</tr>
<tr>
<td valign="top" align="left">No.4</td>
<td valign="middle" align="left"/>
<td valign="bottom" align="left"/>
<td valign="bottom" align="left">&#x221a;</td>
<td valign="top" align="left">92.23</td>
<td valign="top" align="left">90.87</td>
</tr>
<tr>
<td valign="top" align="left">No.5</td>
<td valign="middle" align="left">&#x221a;</td>
<td valign="bottom" align="left">&#x221a;</td>
<td valign="bottom" align="left">&#x221a;</td>
<td valign="top" align="left">94.53</td>
<td valign="top" align="left">92.41</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>(1) No.1 shows the experimental results of the pre-improved YOLOv7 algorithm, which serves as a comparative benchmark for the experiments of the latter 4 groups, detecting an mAP value of 91.10% and an F1 value of 90.20.</p>
</fn>
<fn>
<p>(2) No.2 is to add only the attention mechanism, although the attention mechanism increases the amount of computation and the number of parameters, the mAP value is improved by 1.40%, and the F1 value is improved by 0.93.</p>
</fn>
<fn>
<p>(3) No.3 is to replace only some of the convolutional layers in the backbone network with deformable convolution, the mAP value is improved by 1.20%, and the F1 value is improved by 0.88.</p>
</fn>
<fn>
<p>(4) No.4 for replacing only the WIoU loss function, without increasing the number of model parameters and computational effort, the mAP value is improved by 1.13%, and the F1 value is improved by 0.67.</p>
</fn>
<fn>
<p>(5) No.5 shows the experiment of the improved algorithm in this paper, compared with the YOLOv7 algorithm before improvement, the mAP value is improved by 3.43%, F1 value is improved by 2.21, which confirms the validity of each improvement point. "&#x221a;*" indicates that this methodology is used.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Comparison of experimental results and analysis</title>
<p>In order to evaluate the performance of the algorithms, the improved algorithm proposed in this paper is compared with the algorithms such as YOLOv7, YOLOv5, YOLOv8, Faster-R-CNN, and SSD for the detection performance on the dataset, and all the experiments are conducted under the same parameters. The experimental results are shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, the table below, the improved algorithm, the mAP reached 94.53% and the F1 score reached 92.41, which are 3.43% and 2.21 respectively higher than the baseline YOLOv7, compared to the other algorithms in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> with the best overall results. Faster R-CNN and SSD algorithms are affected by the fixed parameters of the anchor frame, which reduces the model detection effect. The YOLOv5 algorithm optimizes the network structure better, and the predicted position regression is more accurate. YOLOv8 has higher accuracy and recall because the C3 structure of the backbone network is replaced by the C2f structure with a richer gradient flow, and the number of channels is adjusted differently for different scale models, which significantly improves the model performance, but it is higher than There is still a certain gap compared with this paper. Comprehensive measurement of different detection algorithms, the improved algorithm in this paper is better, which confirms the effectiveness of the improved algorithm in this paper.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Models</th>
<th valign="top" align="center">mAP(%)</th>
<th valign="top" align="center">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">YOLOv7</td>
<td valign="top" align="left">91.10</td>
<td valign="top" align="left">90.20</td>
</tr>
<tr>
<td valign="top" align="left">YOLOv5</td>
<td valign="top" align="left">92.34</td>
<td valign="top" align="left">91.20</td>
</tr>
<tr>
<td valign="top" align="left">YOLOv8</td>
<td valign="top" align="left">91.41</td>
<td valign="top" align="left">91.00</td>
</tr>
<tr>
<td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="left">76.89</td>
<td valign="top" align="left">76.00</td>
</tr>
<tr>
<td valign="top" align="left">SSD</td>
<td valign="top" align="left">72.48</td>
<td valign="top" align="left">72.00</td>
</tr>
<tr>
<td valign="top" align="left">Improved-YOLOv7</td>
<td valign="top" align="left">94.53</td>
<td valign="top" align="left">92.41</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>Below we discuss the visualization results after the introduction of the attention mechanism. In <xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8</bold>
</xref>, <xref ref-type="fig" rid="f9">
<bold>9</bold>
</xref>, A is the original image of Cane, (B-F) are the heat maps before and after adding the attention mechanism, respectively, and the darker red color in the heat map indicates the larger value. From <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, it can be seen that the model is not accurate enough in predicting the stem nodes during the target detection, the darkest colored part is not the sugarcane stem node, and there is a false detection. From <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, it can be seen that after adding the attention mechanism, the model is more accurate in locating the stem nodes.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Heat map before adding the attention mechanism: <bold>(A)</bold> Original image; <bold>(B-F)</bold> Heat map of each sugarcane stem node of the original model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g008.tif"/>
</fig>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Heat map after adding the attention mechanism: <bold>(A)</bold> Original image; <bold>(B-F)</bold> Heat map of each sugarcane stem node for the improved model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g009.tif"/>
</fig>
<p>Here we discuss the effect of the improved loss function. <xref ref-type="fig" rid="f10">
<bold>Figures&#xa0;10</bold>
</xref>, show the box loss and total loss before and after the improved algorithm, respectively, and the horizontal coordinates are the number of training rounds. It can be directly concluded from the above figures that the improved algorithm has smaller loss values than the original algorithm, and the values of its box loss and total loss are stable at 0.035, 0.045, and From the figure, we can see that although the improved algorithm has a higher loss value than the original YOLOv7 at the beginning of training, the loss value decreases quickly and stabilizes as the number of training rounds increases. Therefore, it shows that the improved algorithm converges faster and has better performance.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Loss. <bold>(A)</bold> Box loss; <bold>(B)</bold> Total loss.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g010.tif"/>
</fig>
<p>In order to compare the detection effect of the algorithm improvement more intuitively, the detection effect of the original YOLOv7 algorithm, other models, and the improved model of this paper is compared with the real labeled frame as the benchmark, and the detection results of the sugarcane stem nodes are shown in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>. It can be seen that in the case of complex background interference, YOLOv7 and other models will have problems such as incomplete detection and misdetection, etc. In this paper, by improving the YOLOv7 model and strengthening the spatial feature extraction ability of the backbone network, the probability of misdetection of the sugarcane stem nodes is significantly reduced. In the case of sugarcane stem nodes with fuzzy edges, unclear contours, and obscured by leaf wrappings, the original YOLOv7 algorithm is prone to miss detection, and the accuracy of sugarcane stem node detection is low. In contrast, the algorithm in this paper, by introducing deformable convolution in the feature fusion network, acquires a larger sensory field and captures more spatial information, and incorporates the attention mechanism in the stem network, which utilizes the multidimensional interaction between channels and space to retain the key information, so that the model can have the ability to differentiate between overlapping targets, and improve the occurrence of leakage detection well. In this paper, the WIoU loss function is introduced to make the model more accurate in judging the position of sugarcane stem nodes and improve detection accuracy. In summary, it shows that the improved model has good applicability in complex environments.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Comparison experiments.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1230517-g011.tif"/>
</fig>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>In this paper, we propose an improved YOLOv7-based sugarcane stem node detection algorithm for dense sugarcane in the natural environment, small target, large number, and easy to obscure and overlap, and verify the effectiveness of the proposed algorithm through experiments. In order to make the feature extraction network adaptive to extract the shape and location features of dense sugarcane stem nodes in the natural environment, a deformable convolutional network is introduced in the module of the original YOLOv7 algorithm, which is used to replace some ordinary convolutional layers to improve the feature extraction ability of the algorithm for dense and occluded sugarcane stem nodes. In the backbone network of YOLOv7, the SimAM attention mechanism is added to improve the network&#x2019;s ability to extract deep and important features. To improve the matching between the prediction frame and the real labeled frame, the WIoU boundary regression loss function is introduced to replace the CIoU loss function in YOLOv7, which can better guide the network learning and improve the accuracy of detection results. The mAP of the improved YOLOv7 algorithm is 94.53%, which is 3.43% higher than the mAP of the YOLOv7 algorithm, with certain robustness and generalization, and 17.64%, 2.19%, 3.12%, compared with the average accuracy of the Faster-RCNN, YOLOv5, YOLOv8, and SSD network models, respectively, 22.05%, providing a new research idea for the intelligent sugarcane harvester to identify stem nodes. The improved YOLOv7 algorithm can improve the detection accuracy of obscured and overlapping sugarcane stem nodes in the natural environment, but the model detection speed decreases while the accuracy is improved; therefore, in future work, it is still necessary to consider how to further optimize the improved algorithm, use a lighter network while ensuring the detection accuracy and improving its generalization ability, to provide new sugarcane stem node recognition in natural environment Detection methods.</p>
</sec>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>Conceptualization, CW and HG; Methodology, CW and HG; Software, HG; Validation, HG, JL, and BH; Formal analysis, YH, and KL; Writing-original manuscript preparation, HG; Writing-review and editing, HG and JL; Supervision, HN, XL, YL, QW. all authors have read and agreed to the published version of the manuscript.</p>
</sec>
</body>
<back>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>This study was supported by the National Natural Science Foundation of China (52165009), Guangxi Science and Technology Major Project (Guike AA22117006), Guangxi Student Innovation and Entrepreneurship Project (202010608120, S202210608274S), and Chongzuo City Science and Technology Project (Chongke 20220621).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Yolov4: Optimal speed and accuracy of object detection</article-title>. <source>arXiv preprint arXiv:2004.10934</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2004.10934</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Ju</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>b). <article-title>Sugarcane stem node recognition in field by deep learning combining data expansion</article-title>. <source>Appl. Sci.</source> <volume>11</volume>, <fpage>8663</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app11188663</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Qiang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>a). <article-title>Sugarcane nodes identification algorithm based on sum of local pixel of minimum points of vertical projection function</article-title>. <source>Comput. Electron. Agric.</source> <volume>182</volume>, <fpage>105994</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2021.105994</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dai</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). &#x201c;<article-title>Deformable convolutional networks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE international conference on computer vision</conf-name>, <publisher-name>IEEE</publisher-name>, <publisher-loc>345 E 47TH St, New York, NY 10017 USA</publisher-loc>. <fpage>764</fpage>&#x2013;<lpage>773</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hou</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Coordinate attention for efficient mobile network design</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>, <publisher-name>IEEE Computer Soc</publisher-name>, <publisher-loc>10662 Los Vaqueros Circle, PO BOX 3014, Los Alamitos, CA 90720-1264 USA</publisher-loc>. <fpage>13713</fpage>&#x2013;<lpage>13722</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Squeeze-and-excitation networks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>, <publisher-name>IEEE Computer Soc</publisher-name>, <publisher-loc>10662 Los Vaqueros Circle, PO BOX 3014, Los Alamitos, CA 90720-1314</publisher-loc>. <fpage>7132</fpage>&#x2013;<lpage>7141</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Localization and test of characteristics distribution for sugarcane internode based on matlab</article-title>. <source>Trans. Chin. Soc. Agric.</source> <volume>44</volume>, <fpage>93</fpage>&#x2013;<lpage>97</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2013.10.016</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Increasing the real-time dynamic identification efficiency of sugarcane nodes by improved yolov3 network</article-title>. <source>Trans. Chin. Soc. Agric. Eng.</source> <volume>35</volume>, <page-range>185&#x2013;191</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.11975/j.issn.1002-6819.2019.23.023</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Recognition and features extraction of sugarcane nodes based on machine vision</article-title>. <source>Trans. Chin. Soc. Agric.</source> <volume>41</volume>, <fpage>190</fpage>&#x2013;<lpage>194</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3969/j.issn.1000-1298.2010.10.039</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Sugarcane node recognition technology based on wavelet analysis</article-title>. <source>Comput. Electron. Agric.</source> <volume>158</volume>, <fpage>68</fpage>&#x2013;<lpage>78</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2019.01.043</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moshashai</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Almasi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Minaei</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Borghei</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Identification of sugarcane nodes using image processing and machine vision technology</article-title>. <source>Int. J. Agric. Res.</source> <volume>3</volume>, <fpage>357</fpage>&#x2013;<lpage>364</lpage>. doi: <pub-id pub-id-type="doi">10.3923/ijar.2008.357.364</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tong</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Wise-iou: Bounding box regression loss with dynamic focusing mechanism</article-title>. <source>arXiv preprint arXiv:2301.10051</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2301.10051</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Yolov7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>7464</fpage>&#x2013;<lpage>7475</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2207.02696</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Ndiluau</surname> <given-names>P. F.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>b). <article-title>Sugarcane stem node detection and localization for cutting using deep learning</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2022.1089961</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>G.</given-names>
</name>
<name>
<surname>He</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>a). <article-title>Study on pear flowers detection performance of yolo-pefl model trained with synthetic target images</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>911473</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2022.911473</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zuo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Eca-net: Efficient channel attention for deep convolutional neural networks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. (<publisher-loc>Seattle, WA, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>) <fpage>11534</fpage>&#x2013;<lpage>11542</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Webb</surname> <given-names>B. S.</given-names>
</name>
<name>
<surname>Dhruv</surname> <given-names>N. T.</given-names>
</name>
<name>
<surname>Solomon</surname> <given-names>S. G.</given-names>
</name>
<name>
<surname>Tailby</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Lennie</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Early and late mechanisms of surround suppression in striate cortex of macaque</article-title>. <source>J. Neurosci.</source> <volume>25</volume>, <fpage>11666</fpage>&#x2013;<lpage>11675</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.3414-05.2005</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J.-Y.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Cbam: Convolutional block attention module</article-title>,&#x201d; in <conf-name>Proceedings of the European conference on computer vision (ECCV)</conf-name>, <publisher-name>Springer-Verlag Berlin Heidelberger Platz 3</publisher-name>, <publisher-loc>D-14197 Berlin, Germany</publisher-loc>. <fpage>3</fpage>&#x2013;<lpage>19</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>R.-Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Simam: A simple, parameter-free attention module for convolutional neural networks</article-title>,&#x201d; in <conf-name>International conference on machine learning (PMLR)</conf-name>, <publisher-name>JMLR-Journal Machine Learning Research</publisher-name>, <publisher-loc>1269 Law St, San Diego, CA, United States</publisher-loc>. <fpage>11863</fpage>&#x2013;<lpage>11874</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ru</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Detection method of wheat spike improved yolov5s based on the attention mechanism</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>993244</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2022.993244</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Qu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Research on present situation and countermeasures for the development of sugarcane mechanization</article-title>. <source>Guangdong Agric. Sci.</source> <volume>39</volume>, <fpage>196</fpage>&#x2013;<lpage>199</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.16768/j.issn.1004-874x.2012.19.027</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Dragon fruit detection in natural orchard environment by integrating lightweight network and attention mechanism</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>1040923</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2022.1040923</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Enhancing geometric factors in model learning and inference for object detection and instance segmentation</article-title>. <source>IEEE Trans. Cybernetics</source> <volume>52</volume>, <fpage>8574</fpage>&#x2013;<lpage>8586</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TCYB.2021.3095305</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>G.</given-names>
</name>
<name>
<surname>He</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A new design of sugarcane seed cutting systems based on machine vision</article-title>. <source>Comput. Electron. Agric.</source> <volume>175</volume>, <fpage>105611</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2020.105611</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Spatial location of sugarcane node for binocular vision-based harvesting robots based on improved yolov4</article-title>. <source>Appl. Sci.</source> <volume>12</volume>, <fpage>3088</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app12063088</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>