<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="review-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2022.868745</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Application of Convolutional Neural Network-Based Detection Methods in Fresh Fruit Production: A Comprehensive Review</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Chenglin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/903540/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Suchun</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1294424/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Yawei</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1626592/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Xiong</surname> <given-names>Juntao</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1448426/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhang</surname> <given-names>Zhaoguo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhao</surname> <given-names>Bo</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Luo</surname> <given-names>Lufeng</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/796345/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lin</surname> <given-names>Guichao</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1246237/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>He</surname> <given-names>Peng</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Faculty of Modern Agricultural Engineering, Kunming University of Science and Technology</institution>, <addr-line>Kunming</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Intelligent Manufacturing Engineering, Chongqing University of Arts and Sciences</institution>, <addr-line>Chongqing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>College of Mathematics and Informatics, South China Agricultural University</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Chinese Academy of Agricultural Mechanization Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>School of Mechatronic Engineering and Automation, Foshan University</institution>, <addr-line>Foshan</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>School of Mechanical and Electrical Engineering, Zhongkai University of Agriculture and Engineering</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<aff id="aff7"><sup>7</sup><institution>School of Electronic and Information Engineering, Taizhou University</institution>, <addr-line>Taizhou</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Gregorio Egea, University of Seville, Spain</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Orly Enrique Apolo-Apolo, University of Seville, Spain; Mohsen Yoosefzadeh Najafabadi, University of Guelph, Canada</p></fn>
<corresp id="c001">&#x002A;Correspondence: Juntao Xiong, <email>xiongjt2340@163.com</email></corresp>
<corresp id="c002">Zhaoguo Zhang, <email>zhaoguozhang@163.com</email></corresp>
<fn fn-type="other" id="fn004"><p>This article was submitted to Technical Advances in Plant Science, a section of the journal Frontiers in Plant Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>868745</elocation-id>
<history>
<date date-type="received">
<day>03</day>
<month>02</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Wang, Liu, Wang, Xiong, Zhang, Zhao, Luo, Lin and He.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wang, Liu, Wang, Xiong, Zhang, Zhao, Luo, Lin and He</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>As one of the representative algorithms of deep learning, a convolutional neural network (CNN) with the advantage of local perception and parameter sharing has been rapidly developed. CNN-based detection technology has been widely used in computer vision, natural language processing, and other fields. Fresh fruit production is an important socioeconomic activity, where CNN-based deep learning detection technology has been successfully applied to its important links. To the best of our knowledge, this review is the first on the whole production process of fresh fruit. We first introduced the network architecture and implementation principle of CNN and described the training process of a CNN-based deep learning model in detail. A large number of articles were investigated, which have made breakthroughs in response to challenges using CNN-based deep learning detection technology in important links of fresh fruit production including fruit flower detection, fruit detection, fruit harvesting, and fruit grading. Object detection based on CNN deep learning was elaborated from data acquisition to model training, and different detection methods based on CNN deep learning were compared in each link of the fresh fruit production. The investigation results of this review show that improved CNN deep learning models can give full play to detection potential by combining with the characteristics of each link of fruit production. The investigation results also imply that CNN-based detection may penetrate the challenges created by environmental issues, new area exploration, and multiple task execution of fresh fruit production in the future.</p>
</abstract>
<kwd-group>
<kwd>computer vision</kwd>
<kwd>deep learning</kwd>
<kwd>convolutional neural network</kwd>
<kwd>fruit detection</kwd>
<kwd>fruit production</kwd>
</kwd-group>
<counts>
<fig-count count="17"/>
<table-count count="6"/>
<equation-count count="7"/>
<ref-count count="201"/>
<page-count count="28"/>
<word-count count="21094"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Fresh fruits in the market are beloved by people because of their enticing aroma and unique flavor. From fruit flowers blooming to fruit grading, every link of fresh fruit production needs to be seriously supervised so that fruits enter the market without economic loss. In recent years, the world agricultural population and labor force have been having a declining trend leading to the urgent need for automation of fresh fruit production (<xref ref-type="bibr" rid="B185">Yuan et al., 2017</xref>). Object detection based on computer vision has been applied to the main link of automatic fresh fruit production such as smart yield prediction, automatic harvesting robots, and intelligent fruit quality grading (<xref ref-type="bibr" rid="B102">Naranjo-Torres et al., 2020</xref>).</p>
<p>A function of ML is to ensure that machines can automatically detect objects accurately. Although ML has been applied in many fields, the ML technology has been developing to achieve efficient detection. The detection performance of traditional ML will not improve with increase in training sample data. The features need to be given artificially for object detection, which is also a disadvantage of traditional ML (<xref ref-type="bibr" rid="B98">Mohsen et al., 2021</xref>). As an intelligent algorithm in the development of ML, DL has significant advantages over traditional algorithms of ML. The detection performance of DL usually improves with increase in the amount of training sample data. DL can automatically extract features of a detected object using network structure. However, DL takes a lot of training time and runs on computers with higher cost configurations compared with traditional ML (<xref ref-type="bibr" rid="B72">Joe et al., 2022</xref>).</p>
<p>Deep learning is a further study on artificial neural networks such as deep belief network (<xref ref-type="bibr" rid="B60">Hinton et al., 2006</xref>), recurrent neural network (<xref ref-type="bibr" rid="B137">Schuster and Paliwal, 1997</xref>), and convolutional neural network (<xref ref-type="bibr" rid="B84">LeCun et al., 1989</xref>). The deep learning algorithm has a similar calculation principle with a mechanism of the visual cortex of animals (<xref ref-type="bibr" rid="B128">Rehman et al., 2019</xref>). The deep learning-based technology has broad applications in many domains due to its superior performance in operation speed and accuracy, for example, in the medical field (<xref ref-type="bibr" rid="B53">Gupta et al., 2019</xref>; <xref ref-type="bibr" rid="B194">Zhao Q. et al., 2019</xref>), in the aerospace field (<xref ref-type="bibr" rid="B31">Dong Y. et al., 2021</xref>), in the transportation sector (<xref ref-type="bibr" rid="B103">Nguyen et al., 2018</xref>), in the agriculture field (<xref ref-type="bibr" rid="B76">Kamilaris and Prenafeta-Bold&#x00FA;, 2018</xref>), and in the biochemistry field (<xref ref-type="bibr" rid="B4">Angermueller et al., 2016</xref>).</p>
<p>A CNN with a convolutional layer and a pooling layer was proposed by <xref ref-type="bibr" rid="B40">Fukushima (1980)</xref>, which was subsequently improved to LeNet (<xref ref-type="bibr" rid="B85">LeCun et al., 1998</xref>), GoogleNet (<xref ref-type="bibr" rid="B146">Szegedy et al., 2015</xref>), ResNet (<xref ref-type="bibr" rid="B57">He et al., 2016</xref>), AlexNet (<xref ref-type="bibr" rid="B81">Krizhevsky et al., 2017</xref>), and so on. With the appearance of R-CNN (<xref ref-type="bibr" rid="B49">Girshick et al., 2014</xref>), CNN-based object detection became a hot research topic on computer vision and digital image processing (<xref ref-type="bibr" rid="B196">Zhao Z. et al., 2019</xref>). Object detection is the coalition of object classification and object location requiring a network to differentiate an object region from the background and accomplish the classification and location of the object. The technique of CNN-based image segmentation using a CNN model to perceive the representative object of each pixel for classifying and locating objects can be performed for object detection tasks. Frequently used image segmentation models are Mask-R-CNN, U-Net (<xref ref-type="bibr" rid="B131">Ronneberger et al., 2015</xref>), SegNet (<xref ref-type="bibr" rid="B9">Badrinarayanan et al., 2017</xref>), DeepLab (<xref ref-type="bibr" rid="B17">Chen et al., 2018</xref>), and so on.</p>
<p>Early fruit image segmentation algorithms use traditional ML algorithms to identify fruit objects by combining shallow characteristics of fruits such as color, texture, and shape, and mainly included threshold segmentation (<xref ref-type="bibr" rid="B112">Pal and Pal, 1993</xref>), DTI (<xref ref-type="bibr" rid="B123">Quinlan, 1986</xref>), SVM (<xref ref-type="bibr" rid="B23">Cortes and Vapnik, 1995</xref>), cluster analysis (<xref ref-type="bibr" rid="B154">Tsai and Chiu, 2008</xref>), and so on. Color traits of fruits are frequently used in fruit detection (<xref ref-type="bibr" rid="B150">Thendral et al., 2014</xref>; <xref ref-type="bibr" rid="B195">Zhao et al., 2016</xref>). Shape, as an outstanding mark of fruits, is applied to fruit segmentation and recognition (<xref ref-type="bibr" rid="B108">Nyarko et al., 2018</xref>; <xref ref-type="bibr" rid="B148">Tan et al., 2018</xref>). In addition, spectral features and depth information are applied in fruit detection (<xref ref-type="bibr" rid="B15">Bulanon et al., 2009</xref>; <xref ref-type="bibr" rid="B109">Okamoto and Lee, 2009</xref>; <xref ref-type="bibr" rid="B45">Gen&#x00E9;-Mola et al., 2019a</xref>; <xref ref-type="bibr" rid="B87">Lin et al., 2019</xref>; <xref ref-type="bibr" rid="B155">Tsoulias et al., 2020</xref>). The above methods can detect fruit objects; however, they have certain limitations of features expression for fruit object detection in a complex environment. CNN-based detection technology has been proved to have a potential in fresh fruit production by many studies (<xref ref-type="bibr" rid="B79">Koirala et al., 2019b</xref>). Models combined with CNN, for example, CNN + SVM (<xref ref-type="bibr" rid="B29">Dias et al., 2018</xref>), CNN + ms-MLP (<xref ref-type="bibr" rid="B11">Bargoti and Underwood, 2017</xref>), fuzzing mask R-CNN (<xref ref-type="bibr" rid="B64">Huang et al., 2020</xref>), faster R-CNN (<xref ref-type="bibr" rid="B43">Gao et al., 2020</xref>), the Alex-FCN model (<xref ref-type="bibr" rid="B163">Wang et al., 2018</xref>), and 3D-CNN (<xref ref-type="bibr" rid="B162">Wang et al., 2020</xref>), have obtained satisfactory detection results in fruit flower detection, fruit recognition, fruit maturity prediction, and surface defect detection-based fruit grading. These successful studies imply that CNN-based methods can break the technical bottleneck in detection and accelerate the mechanization of fresh fruit production.</p>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, this review investigates the CNN-based detection application in the process of fresh fruit production, which is a complete process from fruit flower detection, growing fruit detection, fruit picking to fruit grading. We provide a comprehensive introduction and analysis of the CNN model and its improved models in fresh fruit production. In addition, different CNN-based detection methods are compared and summarized in each link of fresh fruit production. The arrangement of this article is as follows: Section &#x201C;Common Models and Algorithms of Convolutional Neural Network&#x201D; introduces the composition and algorithms of CNN; Section &#x201C;Implementation Process of Convolutional Neural Network-Based Detection&#x201D; explains the CNN-based detection implementation process; Section &#x201C;Convolutional Neural Network-Based Fresh Fruit Detection&#x201D; investigates the current research on CNN applications in each link of fresh fruit production; Section &#x201C;Challenges and Future Perspective&#x201D; discusses difficulties that will be encountered by CNN-based detection in future research on fresh fruit production; Section &#x201C;Conclusion&#x201D; presents an entire summary of this investigation.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Convolutional neural network (CNN)-based detection application in main links of fresh fruit production.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g001.tif"/>
</fig>
</sec>
<sec id="S2">
<title>Common Models and Algorithms of Convolutional Neural Network</title>
<sec id="S2.SS1">
<title>Convolutional Neural Network Models for Image Detection</title>
<p>Common CNN models used for image detection are usually composed of convolutional layers, activation functions, pooling layers, and full-connected layers (<xref ref-type="bibr" rid="B98">Mohsen et al., 2021</xref>). A CNN model transforms an image into high dimension information, so a computer can read and extract features from the image. In two-dimensional (2D) convolution operation, each pixel value of an input image entering into a convolutional layer is convoluted with a kernel to generate a feature map. When an input image is three-dimensional (3D) or four-dimensional (4D), a multi-dimension convolution operation will be implemented. In the multi-dimension convolution operation, the channel number of kernels is equal to the channel number of input images, and the channel number of output feature maps is the number of kernels (<xref ref-type="bibr" rid="B2">Alzubaidi et al., 2021</xref>). However, in convolutional layers and full-connected layers, the linear connection between the input and the output restricts the ability of a CNN model to solve more complex problems. The activation function is added after the operations of convolution layers and full-connected layers, which can capacitate a CNN model to solve non-linear problems. Common activation functions include the Sigmoid function, the Tanh function, the ReLU function, SoftMax, and so on.</p>
<p>LeNet is the first improved CNN; however, it has not been widely promoted and applied because of simple network structure (<xref ref-type="bibr" rid="B83">LeCun and Bengio, 1995</xref>). AlexNet is the first deep CNN architecture and the first CNN model trained on GPU (<xref ref-type="bibr" rid="B81">Krizhevsky et al., 2017</xref>). A VGG model with four network structures and different configurations was proposed by the Visual Geometry Group of Oxford University in 2014 (<xref ref-type="bibr" rid="B140">Simonyan and Zisserman, 2014</xref>). The most popular network among VGG models is VGG-16 containing thirteen convolutional layers and three full-connected layers. GoogLeNet was a new deep learning structure proposed in 2014 (<xref ref-type="bibr" rid="B146">Szegedy et al., 2015</xref>). The most unique of GoogLeNet is the inception component, which utilizes partial connection to accomplish parameter reduction and computation simplicity. A series of inception components including InceptionV2, InceptionV3, and InceptionV4, was proposed for optimizing GoogLeNet (<xref ref-type="bibr" rid="B147">Szegedy et al., 2016</xref>). By proving the existence of degradation of CNN while its depth is increasing, ResNet was proposed to improve the CNN by designing residual components with the shortcut connection (<xref ref-type="bibr" rid="B57">He et al., 2016</xref>). DenseNet was proposed in 2017, and dense block was the highlight of DenseNet by building connections of all layers with each other to ensure maximum information flow among the layers (<xref ref-type="bibr" rid="B62">Huang et al., 2017</xref>). With the popularization of CNN models, it is required that CNN-based image recognition tasks are implemented on mobile terminals or embedded devices. As a lightweight model, MobileNet was designed to run on the CPU platform, and it had good detection accuracy (<xref ref-type="bibr" rid="B61">Howard et al., 2017</xref>). These models are fundamentals of CNN-based object detection and can help computers learn more information about images because of functions of feature recognition and extraction. The structure and image detection performance of the above common CNN models are summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Structure and performance of common convolutional neural network (CNN) models for image detection.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">CNN models</td>
<td valign="top" align="center">Weight layers</td>
<td valign="top" align="center">Convolution<break/> layer</td>
<td valign="top" align="left">Kernel size</td>
<td valign="top" align="center">Active function</td>
<td valign="top" align="center">Dropout<xref ref-type="table-fn" rid="t1fna"><sup>a</sup></xref></td>
<td valign="top" align="center">LRN<xref ref-type="table-fn" rid="t1fnb"><sup>b</sup></xref></td>
<td valign="top" align="center">BN<xref ref-type="table-fn" rid="t1fnc"><sup>c</sup></xref></td>
<td valign="top" align="center">Top-5 error (on ImageNet)</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AlexNet</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">5</td>
<td valign="top" align="left">3&#x00D7;3, 5&#x00D7;5, 11&#x00D7;11</td>
<td valign="top" align="center">ReLU</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">16.4%</td>
</tr>
<tr>
<td valign="top" align="left">VGG</td>
<td valign="top" align="center">19</td>
<td valign="top" align="center">16</td>
<td valign="top" align="left">3&#x00D7;3</td>
<td valign="top" align="center">ReLU</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">7.3%</td>
</tr>
<tr>
<td valign="top" align="left">GoogleNet<break/> (Inception-V1)</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">21</td>
<td valign="top" align="left">1&#x00D7;1, 3&#x00D7;3, 5&#x00D7;5, 7&#x00D7;7</td>
<td valign="top" align="center">ReLU</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">6.7%</td>
</tr>
<tr>
<td valign="top" align="left">ResNet</td>
<td valign="top" align="center">152</td>
<td valign="top" align="center">151</td>
<td valign="top" align="left">1&#x00D7;1, 3&#x00D7;3, 7&#x00D7;7</td>
<td valign="top" align="center">ReLU</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">3.57%</td>
</tr>
<tr>
<td valign="top" align="left">DenseNet</td>
<td valign="top" align="center">265</td>
<td valign="top" align="center">264</td>
<td valign="top" align="left">1&#x00D7;1, 3&#x00D7;3, 7&#x00D7;7</td>
<td valign="top" align="center">ReLU</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">5.29%</td>
</tr>
<tr>
<td valign="top" align="left">MobileNet</td>
<td valign="top" align="center">28</td>
<td valign="top" align="center">27</td>
<td valign="top" align="left">1&#x00D7;1, 3&#x00D7;3</td>
<td valign="top" align="center">ReLU</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center"><xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t1fna"><p><italic><sup>a</sup>Dropout is a training trick, which means that neural network units are temporarily discarded from the network according to a certain probability in the training process of a deep learning network.</italic></p></fn>
<fn id="t1fnb"><p><italic><sup>b</sup>LRN, local response normalization, is a training trick that can enhance the generalization ability of a model. It creates a competitive mechanism for activities of local neurons, which can make the value of neurons with large responses larger and inhibit neurons with small feedback.</italic></p></fn>
<fn id="t1fnc"><p><italic><sup>c</sup>BN, batch normalization, normalizes the data of each layer and performs linear transformation to improve data distribution.</italic></p></fn>
<fn id="t1fns1"><p><italic>&#x002A;Means that we have not found relevant data about Mobilenet in the public references.</italic></p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="S2.SS2">
<title>Convolutional Neural Network Models for Three-Dimensional Point Cloud Detection</title>
<p>With the development of vision technology, sensors that directly acquire 3D data are becoming more common in robotics, autonomous driving, and virtual/augmented reality applications. Because depth information can eliminate a lot of segmentation ambiguities in 2D images and provides important geometric information, the ability to directly process 3D data is invaluable in these applications. However, 3D data often come in the form of point clouds. Point clouds are typically represented by a set of 3D points that are not arranged in order, each with or without additional features (such as RGB color information). Because of the disordered nature of point clouds and the fact that they are arranged differently from regular mesh-like pixels in 2D images, traditional CNNs struggle to handle this disordered input.</p>
<p>At present, the deep learning point cloud target recognition method mainly has three kinds of point cloud target recognition methods based on views (<xref ref-type="bibr" rid="B75">Kalogerakis et al., 2017</xref>), voxels (<xref ref-type="bibr" rid="B130">Riegler et al., 2016</xref>), and point clouds (<xref ref-type="bibr" rid="B121">Qi et al., 2017a</xref>). Among them, the idea based on views is still to convert three-dimensional data into a two-dimensional representation; that is, 3D data are projected according to different coordinates and different perspectives to obtain a two-dimensional view, and then the two-dimensional image convolution processing method is used to extract features from each view and, finally, aggregate the features to obtain classification and segmentation results. The idea based on voxels is to put an unordered point cloud into the voxel grid, so that it becomes a three-dimensional grid regular data structure, and then as network input data. However, in order to solve problems of view-based and voxel-based computational complexity and information loss, researchers began to consider directly inputting raw point cloud data into the network for processing.</p>
<p>At Stanford University in the United States, <xref ref-type="bibr" rid="B121">Qi et al. (2017a)</xref> proposed a new type of neural network, PointNet, for point cloud identification and segmentation directly using a point cloud as the input object, the spatial transformation network T-Net to ensure the displacement invariance of the input point, a shared multilayer perceptron (MLP) to learn the characteristics of each point, and, finally, the maximum pooling layer to aggregate global features. However, PointNet cannot learn the relationship characteristics between different points in the local neighborhood, and then <xref ref-type="bibr" rid="B122">Qi et al. (2017b)</xref> proposed PointNet++ to improve PointNet, according to the idea of two-dimensional convolution proposed hierarchical point cloud feature learning for local areas, which is composed of sampling layer, grouping layer and feature extraction layer (PointNet) in the hierarchical module, while improving the stability of the network architecture and the ability to obtain details. Later, the description ability of local features was enhanced in order to make the local structure information between points, such as distance and direction, be able to learn in the network.</p>
<p>PointNet inputs an irregular point cloud directly into the deep convolutional network, the framework represents the point cloud as a set of 3D points { (<italic>P</italic>&#x2014;<italic>i</italic> = l, &#x2026;, <italic>n</italic>}, where each point <italic>P</italic> is its 3D coordinates plus additional feature channels such as color, normal vector, and other information; the architecture is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. In response to the point cloud disorder problem, PointNet pointed out that a symmetric method is used; that is, maximum pooling, no matter how many orders there are in <italic>N</italic> points, the maximum eigenvalue in the pooling window corresponding to <italic>N</italic> points is selected for each dimension of the final high-latitude feature and fused into the global feature. For the rotation invariance problem of point cloud, PointNet points out that spacial transform network (STN) is used to solve it. Through the T-Net network to learn the point cloud itself attitude information to obtain a DD rotation matrix (D represents the characteristic dimension), PointNet in the input space transformation using 3&#x00D7;3, feature space transformation using 64&#x00D7;64 to achieve the most effective transformation for the target.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Structure of PointNet.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g002.tif"/>
</fig>
</sec>
<sec id="S2.SS3">
<title>Convolutional Neural Network-Based Detection Algorithms</title>
<p>Convolutional neural network-based detection algorithms mainly include object detection algorithms, semantic segmentation algorithms, and instance segmentation algorithms, which are described in detail as follows.</p>
<sec id="S2.SS3.SSS1">
<title>Object Detection Algorithms</title>
<p>As a kind of object detection algorithm, a two-stage detector is mainly composed of a region proposal generator and classes and bounding box prediction. The R-CNN series is the most representative two-stage detector and includes R-CNN (<xref ref-type="bibr" rid="B49">Girshick et al., 2014</xref>), Fast-R-CNN (<xref ref-type="bibr" rid="B48">Girshick, 2015</xref>), Faster-R-CNN (<xref ref-type="bibr" rid="B129">Ren et al., 2017</xref>), etc. R-CNN is the pioneer in using deep learning for object detection. After that, researchers proposed Fast-R-CNN and Faster-R-CNN in succession to update detection performance. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the structure of Faster-R-CNN, which is frequently used. Besides the above object detection algorithms, R-FCN and Libra R-CNN are also two-stage detectors.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Faster-R-CNN structure. The feature map is extracted by a convolutional neural network, and then the RPN (region proposal network) generates several accurate region proposals according to the feature map. The region proposals are mapped to the feature map. The ROI (region of interest) pooling layer is responsible for collecting proposal boxes and calculating proposal feature maps. Finally, the category of each proposal is predicted through the FC (full connect) layer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g003.tif"/>
</fig>
<p>Compared with a two-stage detector, a one-stage detector conducts classification and bounding box regression after feature extraction without generation of proposal regions. Prediction of objects depends on doing dense sampling on an input picture. Representative one-stage detectors are the YOLO series and SSD (single shot multibox detector). The YOLO series contains YOLOv1 (<xref ref-type="bibr" rid="B124">Redmon et al., 2016</xref>), YOLOv2 (<xref ref-type="bibr" rid="B125">Redmon and Farhadi, 2017</xref>), YOLOv3 (<xref ref-type="bibr" rid="B126">Redmon and Farhadi, 2018</xref>), and YOLOv4 (<xref ref-type="bibr" rid="B13">Bochkovskiy et al., 2020</xref>). Notably, during the evolution of YOLO, a new convolution neural net, DarkNet, was constructed for feature extraction. Furthermore, YOLOv2 referenced the anchor conception from Faster-R-CNN. YOLOv3 contains three different output nets that can predict multi-scale pictures. SSD (<xref ref-type="bibr" rid="B93">Liu W. et al., 2016</xref>) is also a kind of one-stage detector that can implement multi-box prediction. VGG-16 was used as a backbone in SSD. With the development of DL, more improved one-stage detection algorithms have been designed.</p>
<p>A comparison of CNN models between two-stage detectors and one-stage detectors is shown in <xref ref-type="table" rid="T2">Table 2</xref>. As can be seen in <xref ref-type="table" rid="T2">Table 2</xref>, frames per second (FPS) of the one-stage detector are bigger than those of the two-stage detector, which implies that the detection speed of the one-stage detector is faster than that of the two-stage detector. The FPS and mAP of the Mask-R-CNN model are bigger than those of other models of the two-stage detector. It shows that the Mask-R-CNN model has faster detection speed and higher detection accuracy than the two-stage detector. However, in the one-stage detector, no CNN model has faster detection speed and higher detection accuracy. Because of lack of mAP in some CNN models on data of VOC2012 and COCO, the accuracy of the two detectors cannot be compared.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Summary of common CNN-based object detection models.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Type</td>
<td valign="top" align="left">Name</td>
<td valign="top" align="left">Backbone</td>
<td valign="top" align="left">Bounding boxes generation</td>
<td valign="top" align="left">Additional blocks</td>
<td valign="top" align="center">FPS<xref ref-type="table-fn" rid="t2fna"><sup>a</sup></xref></td>
<td valign="top" align="center" colspan="2">mAP/%<hr/></td>
<td valign="top" align="left">References</td>
</tr>
<tr>
<td/>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left"/><td/>
<td valign="top" align="center">VOC2012<xref ref-type="table-fn" rid="t2fnb"><sup>b</sup></xref></td>
<td valign="top" align="center">COCO<xref ref-type="table-fn" rid="t2fnc"><sup>c</sup></xref></td>
<td valign="top" align="left"/></tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Two-stage</td>
<td valign="top" align="left">R-CNN</td>
<td valign="top" align="left">AlexNet</td>
<td valign="top" align="left">SS<xref ref-type="table-fn" rid="t2fnd"><sup>d</sup></xref></td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">0.03</td>
<td valign="top" align="center">59.2</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B49">Girshick et al., 2014</xref></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Fast-R-CNN</td>
<td valign="top" align="left">VGG-16</td>
<td valign="top" align="left">SS+ROI pooling</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">7.00</td>
<td valign="top" align="center">68.4</td>
<td valign="top" align="center">19.7</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B48">Girshick, 2015</xref></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="left">VGG-16/ResNet-101</td>
<td valign="top" align="left">RPN+ROI pooling</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">7.00/5.00</td>
<td valign="top" align="center">70.4/73.8</td>
<td valign="top" align="center">21.9/34.9</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B129">Ren et al., 2017</xref></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="left">ResNeXt-101-FPN</td>
<td valign="top" align="left">RPN+ROI align</td>
<td valign="top" align="left">FCN</td>
<td valign="top" align="center">11.00</td>
<td valign="top" align="center">73.9</td>
<td valign="top" align="center">39.8</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B55">He K. et al., 2017</xref></td>
</tr>
<tr>
<td valign="top" align="left">One-stage</td>
<td valign="top" align="left">SSD</td>
<td valign="top" align="left">VGG-16</td>
<td valign="top" align="left">Anchor</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">19.3</td>
<td valign="top" align="center">78.5</td>
<td valign="top" align="center">28.8</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B93">Liu W. et al., 2016</xref></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">YOLOv1</td>
<td valign="top" align="left">GoogleNet</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">45.0</td>
<td valign="top" align="center">57.9</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B124">Redmon et al., 2016</xref></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">YOLOv2</td>
<td valign="top" align="left">DarkNet-19</td>
<td valign="top" align="left">Anchor</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">40.0</td>
<td valign="top" align="center">73.5</td>
<td valign="top" align="center">21.6</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B125">Redmon and Farhadi, 2017</xref></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">YOLOv3</td>
<td valign="top" align="left">DarkNet-53</td>
<td valign="top" align="left">Anchor</td>
<td valign="top" align="left">FPN, SPP</td>
<td valign="top" align="center">51.0</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">33.0</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B126">Redmon and Farhadi, 2018</xref></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">YOLOv4</td>
<td valign="top" align="left">CSPDarkNet53</td>
<td valign="top" align="left">Anchor</td>
<td valign="top" align="left">FPN+PA, SPP</td>
<td valign="top" align="center">23.0</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">43.5</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B13">Bochkovskiy et al., 2020</xref></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t2fna"><p><italic><sup>a</sup>FPS, frames per second, is used to measure how many frames (pictures) the target network can detect per second.</italic></p></fn>
<fn id="t2fnb"><p><italic><sup>b</sup>VOC2012: a dataset used in pattern analysis, statistical modeling, and computational learning visual object classes challenge 2012.</italic></p></fn>
<fn id="t2fnc"><p><italic><sup>c</sup>COCO: Microsoft Common Objects in Context, a dataset funded and labeled by Microsoft in 2014.</italic></p></fn>
<fn id="t2fnd"><p><italic><sup>d</sup>SS: selective search (<xref ref-type="bibr" rid="B158">Uijlings et al., 2013</xref>).</italic></p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="S2.SS3.SSS2">
<title>Semantic Segmentation Algorithms</title>
<p>Unlike box recognition in object detection, semantic segmentation refers to pixel-level recognition and classification, which classifies pixels of the same class into one group. Early DL-based semantics segmentation methods performed clustering to generate super-pixels and a classifier to classify them (<xref ref-type="bibr" rid="B24">Couprie et al., 2013</xref>; <xref ref-type="bibr" rid="B33">Farabet et al., 2013</xref>). However, such methods have drawbacks of time-consuming and rough segmentation results. With the popularity and development of object detection algorithms based on CNNs, semantic segmentation algorithms have also made great progress, and can be divided into region-classification-based image semantic segmentation and pixel-classification-based image semantic segmentation.</p>
<p>The method of region-classification-based image semantic segmentation first selects the appropriate region, then classifies the pixels in the candidate region. SDS (simultaneous detection and segmentation) is a model based on R-CNN that can simultaneously detect and semantically segment targets (<xref ref-type="bibr" rid="B54">Hariharan et al., 2014</xref>). In 2016, based on the SDS method, <xref ref-type="bibr" rid="B92">Liu S. et al. (2016)</xref> convoluted images using sliding windows of different sizes and constructed multi-scale feature maps, proposed an MPA (multi-scale patch aggregation) method that can semantically segment an image at the instance level. DeepMask is a segmentation model proposed based on CNN to generate object proposals (<xref ref-type="bibr" rid="B120">Pinheiro et al., 2015</xref>). It generates image patches directly from original image data and then generates a segmentation mask for given image patches. The whole process is applied to a complete image to improve the efficiency of segmentation.</p>
<p>The method of pixel-classification-based semantic segmentation does not need to generate object candidate regions but extracts image features and information from labeled images. Based on that information, a segmentation model can learn and infer the classes of pixels in an original image, and classify each pixel in the image directly to achieve end-to-end semantic segmentation. FCN (fully convolutional network) is a popular semantic segmentation model that can be compatible with any size of images (<xref ref-type="bibr" rid="B138">Shelhamer et al., 2017</xref>). FCN can distinguish the categories of pixels directly, which greatly promotes the development of semantic segmentation. Subsequently, researchers proposed a series of methods based on FCN. FCN-based image semantic segmentation methods are as follows: DeepLab, DeepLab-V2, and DeepLab-V3. Image semantics segmentation methods based on encoder-decoder model are as follows: U-net, Segnet, Deconvnet, and GCN (global convolution network).</p>
</sec>
<sec id="S2.SS3.SSS3">
<title>Instance Segmentation Algorithms</title>
<p>The purpose of instance segmentation is to distinguish different kinds of objects in an image and different instances of the same kind. Therefore, it has the characteristics of object detection and semantic segmentation at the same time. Because of the characteristics of instance segmentation, it can include instance segmentation based on object detection and instance segmentation based on semantics segmentation.</p>
<p>An instance segmentation algorithm based on object detection has been the mainstream direction in the field of instance segmentation research in recent years. Its main process is to locate an instance using an object detection algorithm, and then segment the instance in each detected box. Mask-R-CNN is one of the famous models in instance segmentation proposed by <xref ref-type="bibr" rid="B55">He K. et al. (2017)</xref>. Mask-R-CNN is one of the famous models in instance segmentation on the basis of Fast-R-CNN(<xref ref-type="bibr" rid="B55">He K. et al., 2017</xref>). As a representative instance segmentation model, many scholars are deeply inspired by Mask-R-CNN. Based on Mask-R-CNN, PANet (path aggregation network) introduces a bottom-up path augmentation structure, adaptive feature pooling, and a fully connected fusion structure to obtain more accurate segmentation results (<xref ref-type="bibr" rid="B91">Liu S. et al., 2018</xref>). <xref ref-type="bibr" rid="B17">Chen et al. (2018)</xref> proposed Masklab, which uses directional features to segment instances of the same semantic class. In 2019, the first instance segmentation algorithm based on a one-stage object detection algorithm, YOLACT, was proposed by <xref ref-type="bibr" rid="B14">Bolya et al. (2019)</xref>. It added a mask generation branch behind the one-stage object detector to complete a segmentation task. The overall structure of YOLACT is relatively lightweight, and the trade-off between speed and effect would be good. In addition, there are some newly proposed instance segmentation algorithms such as MS-R-CNN (<xref ref-type="bibr" rid="B65">Huang et al., 2019</xref>), BMask-R-CNN (<xref ref-type="bibr" rid="B21">Cheng et al., 2020</xref>) and BPR (<xref ref-type="bibr" rid="B149">Tang et al., 2021</xref>).</p>
<p>An instance segmentation algorithm based on semantic segmentation classifies each pixel first and then segments different instances of the same category. For example, the SGN (<xref ref-type="bibr" rid="B90">Liu et al., 2017</xref>) model decomposes an instance segmentation into multiple subtasks, then uses a series of neural networks to complete these subtasks, and finally recombines the results of the subtasks to obtain the segmentation task.</p>
</sec>
<sec id="S2.SS3.SSS4">
<title>Differences of Detection Algorithms</title>
<p>In this section, differences among object detection, semantic segmentation, and instance segmentation are visually explained through pear flower detection. <xref ref-type="fig" rid="F4">Figure 4A</xref> is an undetected image of pear flowers. The result of detecting pear flowers with the object detection algorithm is shown in <xref ref-type="fig" rid="F4">Figure 4B</xref>, and it shows the approximate position of pear flowers with bounding boxes. The result with semantic segmentation algorithm is shown in <xref ref-type="fig" rid="F4">Figure 4C</xref>, which reaches the pixel level compared with the result of object detection. It means that when labeling data sets, the annotation of the task of semantic segmentation is also at pixel level. Compared with rectangular box annotation in the object detection task, the annotation of semantic segmentation task is more complex. The result with the instance segmentation algorithm is shown in <xref ref-type="fig" rid="F4">Figure 4D</xref>, and the detection results of instance segmentation are more detailed than those of semantic segmentation in distinguishing each pear flower individual.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Different CNN-based algorithms for pear flower detection. <bold>(A)</bold> Original image, <bold>(B)</bold> object detection, <bold>(C)</bold> semantic segmentation, and <bold>(D)</bold> instance segmentation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g004.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="S3">
<title>Implementation Process of Convolutional Neural Network-Based Detection</title>
<p>This section introduces the main procedures of comprehensively training a CNN-based deep learning model for basic tasks. The first step is determining the learning target and establishing the data set. Second, it is vital to choose an adept deep learning framework to modify the model and implement training. Finally, mastering the estimation metrics of deep learning models leads to knowing the performance of the modified models and training results.</p>
<sec id="S3.SS1">
<title>Data Set Construction</title>
<sec id="S3.SS1.SSS1">
<title>Dataset Acquisition</title>
<p>An RGB camera, which can capture the properties of a fruit surface, such as color, shape, defect, and texture, is a pervasive and affordable camera for image acquisition used in many types of research (<xref ref-type="bibr" rid="B38">Fu et al., 2020a</xref>). <xref ref-type="bibr" rid="B159">Vasconez et al. (2020)</xref> held an RGB camera and acquired apple, avocado, and lemon pictures at 30 frames per second in orchards. However, the information obtained from RGB images is not sufficient for 3D location and reconstruction. Thus, most researchers have begun utilizing RGB-D to capture RGB images and depth images in their experiments. RGB-D cameras generally operate with three depth measurement principles: structured light, time of flight, and active infrared stereo technique (<xref ref-type="bibr" rid="B38">Fu et al., 2020a</xref>). Data sets that provide geometric information and radiation information can enhance the models&#x2019; ability to distinguish fruits from complex environments. <xref ref-type="bibr" rid="B46">Gen&#x00E9;-Mola et al. (2019b)</xref> established an apple data set containing multimodal RGB-D images and pointed out that the model provided with RGB-D images is more robust than that provided with RGB images in a complex environment. However, sensors in most depth cameras cannot obtain information beyond 3.5 m, and light detection and ranging (LiDAR) scanners are needed to acquire information at a far distance (<xref ref-type="bibr" rid="B155">Tsoulias et al., 2020</xref>). A LiDAR scanner can directly provide three-dimensional positioning information of fruits without being affected by light conditions. In addition, LiDAR data can improve the positioning accuracy of fruits because of the appearance of different objects showing different reflectivity to laser. <xref ref-type="bibr" rid="B45">Gen&#x00E9;-Mola et al. (2019a)</xref>, by detecting Fuji apples in orchards with LiDAR, found that the reflection of apple surface was 0.8 higher than that of leaves and branches at a wavelength of 905 nm.</p>
<p>The internal properties of fruits need hyperspectral reflectance images to be represented. <xref ref-type="bibr" rid="B183">Yu et al. (2018)</xref> used a hyperspectral imaging system that constituted of a spectrometer, a CDD camera, a light system, and a computer to detect the internal features of Korla fragrant pear. Some scholars bought a designed hyperspectral system for data collection (<xref ref-type="bibr" rid="B162">Wang et al., 2020</xref>).</p>
</sec>
<sec id="S3.SS1.SSS2">
<title>Data Set Augmentation</title>
<p>Data sets, as an input, play a significant part in a DL model. Most researchers consider that enhancing the scale and quality of data sets can strengthen the models&#x2019; generalization and learning capacity. The methods of dataset augmentation can be divided into the basic-image-manipulation-based method and the DL-based method. The most straightforward and frequently-used methods based on basic image processing are geometric transformations, flipping, color space, cropping, rotation, translation, noise injection, color space transformations, kernel filters, mix images, and random erasing. <xref ref-type="fig" rid="F5">Figure 5</xref> displays example images with some usual image processes. In addition, the DL-based method contains SMOTE (<xref ref-type="bibr" rid="B16">Chawla et al., 2002</xref>), adversarial training, DC-GAN (deep convolutional GAN) (<xref ref-type="bibr" rid="B197">Zheng et al., 2017</xref>), CycleGAN (<xref ref-type="bibr" rid="B201">Zhu et al., 2017</xref>), CVAE-GAN (<xref ref-type="bibr" rid="B10">Bao et al., 2017</xref>), etc.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Example images with different image processes. <bold>(A)</bold> Original image, <bold>(B)</bold> vertical flip image, <bold>(C)</bold> noise injected image, <bold>(D)</bold> sharpened image, <bold>(E)</bold> Gaussian blurry image, <bold>(F)</bold> random erased image, <bold>(G)</bold> image with brightness adjustment, <bold>(H)</bold> RGB2GRB image, and <bold>(I)</bold> gray image.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g005.tif"/>
</fig>
<p>Some researchers processed images from angle, brightness, and sharpness to simulate different light conditions (<xref ref-type="bibr" rid="B68">Jia et al., 2020</xref>). Some used clockwise rotation, horizontal mirror, color balance processing, and blur processing to augment a data set for apple detection (<xref ref-type="bibr" rid="B152">Tian et al., 2019</xref>). Flowers have distinct characteristics from fruit organs. Thus, <xref ref-type="bibr" rid="B151">Tian et al. (2020)</xref> proposed a novel image augmentation method as per apple inflorescence (<xref ref-type="fig" rid="F6">Figure 6</xref>). The procedure of image generation is displayed in <xref ref-type="fig" rid="F7">Figure 7</xref>. They clipped 50 pictures of central flowers and 150 pictures of side flowers. Then, they filtered and combined these clipped images to generate foreground pictures. At the same time, 200 pictures were extracted and processed for background pictures. Finally, sample images were produced by coalescing foreground pictures and background pictures. The experiment results proved that this way of augmentation contributed to detection performance.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Apple inflorescence: <bold>(A)</bold> the central flower and the side flowers have a bud shape, <bold>(B)</bold> the central flower has a semi-open shape and the side flowers have a bud shape, <bold>(C)</bold> the central flower has a fully open shape and the side flowers have bud and semi-open shapes, and <bold>(D)</bold> the central flower and the side flowers have a fully open shape.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Procedure of image generation in <xref ref-type="bibr" rid="B151">Tian et al. (2020)</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g007.tif"/>
</fig>
</sec>
</sec>
<sec id="S3.SS2">
<title>Convolutional Neural Network Model Training</title>
<sec id="S3.SS2.SSS1">
<title>Training Tools</title>
<p>It is onerous to construct a deep learning model from zero. Many open-source or commercial deep learning tools came into being with the advent of deep learning (<xref ref-type="bibr" rid="B86">Li et al., 2021</xref>). In the field of fresh fruit detection, Caffe, TensorFlow, Keras, and PyTorch are popular open-source training tools.</p>
<p>Caffe is the abbreviation of convolution architecture for feature extraction, and is one of the earlier DL frameworks. Caffe defines a network structure in the form of configuration text instead of code. Users can expand new models and learning tasks with its modular components (<xref ref-type="bibr" rid="B69">Jia et al., 2014</xref>). TensorFlow is an open-source machine learning library from Google Brain that can be used for a variety of deep learning tasks, including CNN, RNN, and GAN (generative adversarial network) (<xref ref-type="bibr" rid="B1">Abadi et al., 2016</xref>). It uses data flow graphs to represent calculations, shared states, and operations (<xref ref-type="bibr" rid="B203">Zhu et al., 2018</xref>). Keras is a very friendly and simple DL framework for beginners. Strictly speaking, it is not an open-source framework but a highly modular neural network library based on TensorFlow and Theano. PyTorch is a DL framework launched by Facebook in 2017 and is based on the original Torch framework; it utilizes Python as main development language (<xref ref-type="bibr" rid="B115">Paszke et al., 2019</xref>). Furthermore, the open-source code of Caffe2 has merged into PyTorch, which signifies that PyTorch has strong capacity and flexibility. <xref ref-type="table" rid="T3">Table 3</xref> describes the detail and differences of the above DL tools. In <xref ref-type="table" rid="T4">Table 4</xref>, we display the code of the first convolutional layer of Lenet-5 in different languages.</p>
<table-wrap position="float" id="T3">
<label>TABLE 3</label>
<caption><p>Comparison of Caffe, TensorFlow, Keras, and PyTorch.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Name</td>
<td valign="top" align="left">Caffe</td>
<td valign="top" align="left">TensorFlow</td>
<td valign="top" align="left">Keras</td>
<td valign="top" align="left">PyTorch</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Support language</td>
<td valign="top" align="left">C++/Python/MATLAB</td>
<td valign="top" align="left">C++/Python</td>
<td valign="top" align="left">Python</td>
<td valign="top" align="left">Python</td>
</tr>
<tr>
<td valign="top" align="left">Support hardware</td>
<td valign="top" align="left">CPU/GPU</td>
<td valign="top" align="left">CPU/GPU/Mobile</td>
<td valign="top" align="left">CPU/GPU/Mobile</td>
<td valign="top" align="left">CPU/GPU</td>
</tr>
<tr>
<td valign="top" align="left">Support system</td>
<td valign="top" align="left">Linux/Windows/MacOS</td>
<td valign="top" align="left">Linux/Windows/MacOS/Android/IOS</td>
<td valign="top" align="left">Linux/Windows/MacOS/Android/IOS</td>
<td valign="top" align="left">Linux/Windows/MacOS</td>
</tr>
<tr>
<td valign="top" align="left">Traits</td>
<td valign="top" align="left">Strong readability and expansibility, stable and superior performance</td>
<td valign="top" align="left">Comprehensive functionality, good visualization, and active user community</td>
<td valign="top" align="left">Highly modular, keeping each module short and simple, and ease of extension.</td>
<td valign="top" align="left">Intuitive design, ease of use, and active user community</td>
</tr>
</tbody>
</table></table-wrap>
<table-wrap position="float" id="T4">
<label>TABLE 4</label>
<caption><p>Different languages define the code of the first convolution layer of Lenet-5.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<tbody>
<tr>
<td valign="top" align="left"><inline-graphic xlink:href="fpls-13-868745-t004.jpg"/></td>
</tr>
</tbody>
</table></table-wrap>
<p>Furthermore, data set annotation, which generates ground truth for supervising networks&#x2019; learning object features, is a prerequisite for tasks of object detection and segmentation. Familiar label tools have LabelImg, LabelMe (<xref ref-type="bibr" rid="B133">Russell et al., 2007</xref>), Matlab, Yolo_mark, Vatic, CVAT, etc.</p>
</sec>
<sec id="S3.SS2.SSS2">
<title>Parameter Tuning</title>
<p>Parameter initialization is very important. Reasonable initial parameters can help a model improve training speed and avoid local minima. The Kaiming initialization and Glorot initialization methods are generally used (<xref ref-type="bibr" rid="B50">Glorot and Bengio, 2010</xref>; <xref ref-type="bibr" rid="B56">He et al., 2015</xref>).</p>
<p>In the beginning of the training, all parameters have typically random values and, therefore, far away from the final solution. Using a too-large learning rate may result in numerical instability. We can use warm-up heuristic (<xref ref-type="bibr" rid="B59">He et al., 2019</xref>) to gradually increase the learning rate parameter from 0 to the initial learning rate, and then use the conventional learning rate attenuation scheme. With the progress of training, a model will gradually converge to the global optimum. It is necessary to reduce the learning rate to prevent a model from oscillating back and forth near the optimum. Generally, learning rate adjustment strategies such as Step, MultiStep, and exponential and cosine annealing can be used.</p>
<p>Selection of an optimizer plays an important role in DL training and is related to whether the training can converge quickly and achieve high accuracy and recall. Commonly used optimizers include gradient descent, momentum, SGD, SGDM, Adagrad, Rmsprop, Adam, etc.</p>
<p>Convolutional neural network learning needs to establish millions of parameters and a large number of labeled images. If the amount of data is not enough, a model will be over fitted, and the effect is likely to be worse than traditional manual features. If the data set of a new task is significantly different from the original data set and the amount of data is small, one can try transfer learning to complete the new task (<xref ref-type="bibr" rid="B111">Oquab et al., 2014</xref>). The weight update of a whole network can be adopted during transfer learning.</p>
</sec>
</sec>
<sec id="S3.SS3">
<title>Evaluation Metrics</title>
<p>The confusion matrix is a basic, intuitive, computational, and simple method for measuring the accuracy of a model. Take the binary classification model as an example, and its confusion matrix is shown in <xref ref-type="fig" rid="F8">Figure 8</xref>. It is mainly composed of four basic indicators: TP (true positive), FN (false negative), FP (false positive), and TN (true negative).</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption><p>Basic confusion matrix.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g008.tif"/>
</fig>
<list list-type="simple">
<list-item>
<label>&#x2022;</label>
<p>TP: an outcome where a model correctly predicts a positive class.</p>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>FP: an outcome where a model incorrectly predicts a positive class.</p>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>TN: an outcome where a model correctly predicts a negative class.</p>
</list-item>
<list-item>
<label>&#x2022;</label>
<p>FN: an outcome where a model incorrectly predicts a negative class.</p>
</list-item>
</list>
<p>With a confusion matrix, accuracy, precision, recall, and F1-score can be calculated to evaluate a model. Accuracy (Eq. 1) indicates the proportion of correctly classified test instances to the total number of test instances. Precision (Eq. 2) represents the correct proportion of positive samples predicted by a model. Recall (Eq. 3) represents the proportion of all positive samples that are correctly predicted by a model. Generally speaking, precision and recall is a pair of contradictory indicators. As the weighted harmonic average of the two of them, F1-score (Eq. 4) balances the relative importance between precision and recall.</p>
<disp-formula id="S3.E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S3.E2">
<label>(2)</label>
<mml:math id="M2">
<mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S3.E3">
<label>(3)</label>
<mml:math id="M3">
<mml:mrow>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S3.E4">
<label>(4)</label>
<mml:math id="M4">
<mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In addition to the above basic evaluation metrics, there are also IoU (intersection over union) and mAP (mean average precision) for evaluating the accuracy of a bounding box in an object detection and segmentation model, FPS for detection of speed, and the metrics of the regression model of MAE (mean absolute error), MSE (mean square error), RMSE (root mean square error), and <italic>R</italic><sup>2</sup> coefficient of determination, etc. Diversified evaluation indicators can help researchers evaluate and improve algorithms used in many aspects.</p>
<p>ROC curve is often used for evaluating two classifiers. The vertical axis of the ROC diagram is TPrate (Eq. 5) and the horizontal axis is FPrate (Eq. 6). FPrate represents the probability of misclassifying negative cases into positive cases, and TPrate represents the probability that positive cases can be divided into pairs. Each discrete classifier produces an (FPrate, TPrate) pair corresponding to a single point in ROC space. Several points in the ROC space are important to note. The lower left point (0, 0) represents the strategy of never issuing a positive classification; such a classifier commits no false positive errors but also gains no true positives. The opposite strategy of unconditionally issuing positive classifications is represented by the upper right point (1, 1). The point (0, 1) represents perfect classification (<xref ref-type="bibr" rid="B35">Fawcett, 2006</xref>).</p>
<disp-formula id="S3.E5">
<label>(5)</label>
<mml:math id="M5">
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S3.E6">
<label>(6)</label>
<mml:math id="M6">
<mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In addition to ROC curve, MCC (Eq. 7) is also used to measure the performance of binary classification. This indicator considers true positives, true negatives, false positives, and false negatives. It is generally considered to be a relatively balanced indicator. It can be applied even when sample sizes of two categories are very different (<xref ref-type="bibr" rid="B145">Supper et al., 2007</xref>). MCC is essentially a correlation coefficient between actual classification and prediction classification, and its value range is [&#x2212;1, 1]. When it is 1, it means perfect prediction of a subject; when it is 0, it means that the predicted result is worse than the random prediction result; &#x2212;1 means that the predicted classification is completely inconsistent with the actual classification.</p>
<disp-formula id="S3.E7">
<label>(7)</label>
<mml:math id="M7">
<mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mtext>CC</mml:mtext>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>TN</mml:mi>
</mml:mrow>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:mi>FP</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FP</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FN</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>TN</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FP</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>TN</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FN</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
</sec>
<sec id="S4">
<title>Convolutional Neural Network-Based Fresh Fruit Detection</title>
<sec id="S4.SS1">
<title>Fruit Flower Detection</title>
<p>Fruit flowers are the primary form of fruit organ. Most fruit trees bloom far more than final fruits. However, if there are too many flowers, nutrition supply will be insufficient, which will not only affect the normal development of fruits but will also cause formation of many small fruits and secondary fruits. Yield and economic benefits will be affected. Therefore, flower thinning is necessary to remove some excessive flowers and obtain high-quality fruits (<xref ref-type="bibr" rid="B167">Wouters et al., 2012</xref>). After flower thinning, flower detection is implemented and plays a considerable role in fresh fruit production. Flowers of most kinds of fruits are small and dense, resulting in overlap and blockage, which seriously affect the accuracy of detection. Precise estimation based on DL can assist orchardists in assigning labor resources on time to attain a highly effective but low-cost harvest.</p>
<p>The size of flowers of most species of fruits is small, and the flowers are dense, which causes overlap and occlusion quickly. Many researchers detect the flowers in outdoor fields close to make the most of flowers&#x2019; traits. Being inspired by the performance of CNNs in computer vision tasks, <xref ref-type="bibr" rid="B29">Dias et al. (2018)</xref> incorporated CNN and SVM for apple flower detection. <xref ref-type="bibr" rid="B89">Lin et al. (2020)</xref> compared the performance of R-CNN, Fast-R-CNN, and Faster-R-CNN in recognizing strawberry flowers, and Faster-R-CNN ha higher accuracy (86.1%) than R-CNN (63.4%) and Fast-R-CNN (76.7%). <xref ref-type="bibr" rid="B34">Farjon et al. (2020)</xref> constructed a system for apple flower detection, density calculation, and flourish peak prediction. The detector in the system was based on Faster-R-CNN. Mask R-CNN with ResNeXt50 is a superior algorithm for recognizing citrus flowers and detecting their quality in an end-to-end model. The average precision of detecting citrus flowers is 36.3, and the error of calculating the number was decreased to 11.9% (<xref ref-type="bibr" rid="B28">Deng et al., 2020</xref>). Using U-Net (<xref ref-type="bibr" rid="B131">Ronneberger et al., 2015</xref>) as the backbone of Mask-Scoring-R-CNN can also detect flowers with great precision (<xref ref-type="bibr" rid="B151">Tian et al., 2020</xref>). At the same time, researchers augmented a data set based on apple flowers&#x2019; growth and distribution features to improve the learning capacity of networks. YOlOv4 can detect objects on three different scales. <xref ref-type="bibr" rid="B169">Wu D. et al. (2020)</xref> proposed a channel-pruning algorithm based on the YOLOv4 model. The pruned model contains simple structures and has fewer parameters, and it works with sound accuracy and faster speed.</p>
<p>Grape flower counting is often very time-consuming and laborious because the grape flower has particular phenotypic traits that their shapes are the small sphere and growing on the inflorescence densely. Hence, scholars utilized full convolution net (FCN) to detect and identify inflorescences, and then used CHT to recognize the flowers (<xref ref-type="bibr" rid="B132">Rudolph et al., 2019</xref>). <xref ref-type="bibr" rid="B113">Palacios et al. (2020)</xref> also detected inflorescences and flowers, but both steps used the SegNet architecture with a VGG19 network. In addition, they estimated the actual number of flowers from the number of detected flowers by training a linear regression model. Litchi flowers are also densely clustered and difficult to distinguish in morphology. Thus, a semantic segmentation net that constituted of a backbone net, DeepV3, for feature extraction and a full convolutional net for pixel prediction can detect litchi flower at the pixel level (<xref ref-type="bibr" rid="B173">Xiong et al., 2021</xref>).</p>
</sec>
<sec id="S4.SS2">
<title>Growing Fruit Detection</title>
<sec id="S4.SS2.SSS1">
<title>Terrestrial Platform</title>
<p>In addition to fruit flower detection, fruit detection and counting are also important for yield estimation. Fruit growth in fruit trees is different, and fruit thinning needs to be implemented to remove small fruits, residual fruits, diseased fruits, and fruits with incorrect shapes, so that fruits are evenly distributed in trees and branches and can fully receive nutrients. After the fruit thinning and fruit dropping stages, fruits can be detected during fruit ripening to estimate yield (<xref ref-type="bibr" rid="B198">Zhou et al., 2012</xref>).</p>
<p>The CNN algorithm has better performance for detecting expanding fruits in a vast scene, which has been proved by comparing it with existing methods (<xref ref-type="bibr" rid="B11">Bargoti and Underwood, 2017</xref>). Various species of fruits have different characteristics; therefore, different CNN models are used. <xref ref-type="bibr" rid="B156">Tu et al. (2020)</xref> proposed a MS-FRCNN model to estimate passion fruit production. To detect fruits of small and dense olive, researchers tested five different CNN configurations in an intensive olive orchard, and the model with Inception-ResNetV2 showed the best behavior (<xref ref-type="bibr" rid="B6">Aquino et al., 2020</xref>). <xref ref-type="bibr" rid="B12">Behera et al. (2021)</xref> proposed a Faster-R-CNN model with MIoU, and it achieved an F1 score of 0.9523 and 0.9432 for yield estimation of apple and mango in the ACFR data set. <xref ref-type="bibr" rid="B67">Janowski et al. (2021)</xref> employed the YOLOv3 network to predict the yield of an apple orchard. Nevertheless, all algorithms face the problems of occlusion resulting from leaves or branches and fruit overlap. To suppress the disturbance from occlusion, an instance segmentation neural net based on Mask-R-CNN was used to detect apples in two-dimensional space and a multi-view structure from motion (SFM) (<xref ref-type="bibr" rid="B153">Triggs et al., 2002</xref>) was used to generate a 3D point cloud according to 2D detection results. Recognizing unripe tomatoes is important for long-term yield prediction, but green fruits are hard to perceive in a green background. <xref ref-type="bibr" rid="B101">Mu et al. (2020)</xref> used Faster-R-CNN to detect immature tomatoes in greenhouses and created a tomato location map from detected images. Prediction errors of a whole orchard caused by duplicate statistics attracted the attention of many scholars. It is remarkably effective segmenting individual mango trees with LiDAR Mask and identifying fruits with a Faster-R-CNN-based detector. <xref ref-type="bibr" rid="B80">Koirala et al. (2019a)</xref> designed a mango identification system and installed it on a multifunctional agricultural car to realize real-time detection. The algorithm named &#x201C;MnagoYOLO&#x201D; in the detction system is modified based on YOLOv2. The car drove on the path between rows of mango trees while the system detected and summed the mangoes on the trees (<xref ref-type="bibr" rid="B80">Koirala et al., 2019a</xref>). Some researchers thought of using mobile phones to detect kiwifruits in an orchard in real-time (<xref ref-type="bibr" rid="B200">Zhou et al., 2020</xref>). They used a single shot multi-box detector (SSD) with two lightweight backbones, MobileNetV2 and InceptionV3, to develop a device for kiwifruit detection in the wild, the Android app KiwiDetector. Four types of smart phones are used for experiments. Highest detection accuracy can reach 90.8%, and fastest detection speed can reach 103 ms.</p>
<p>Deep learning has advantages in yield estimation of clustered fruits. For dense small fruits such as blueberries and small tomatoes, DL has a better detection effect on single fruits and is more convenient for counting fruits. However, using DL to detect small fruits is more vulnerable to the influence of light conditions. To quantify the number of berries per image, a network based on Mask R-CNN for object detection and instance segmentation was proposed by <xref ref-type="bibr" rid="B52">Gonzalez et al. (2019)</xref>. Grapes are a type of crop presenting a large variability in phenotype. <xref ref-type="bibr" rid="B187">Zabawa et al. (2020)</xref> chose to train a CNN to implement semantic segmentation for single grape berry detection, and then used the connected component algorithm to count each berry. SfM (structure-from-motion) can simultaneously solve camera pose and scene geometry estimation to find a three-dimensional structure. Thus, <xref ref-type="bibr" rid="B136">Santos et al. (2020)</xref> used Mask-R-CNN to segment grape clusters and generate comprehensive instance masks. Then, the COLMAP SfM software can match and track these masks to reduce duplicate statistics. GPS was employed to establish pair-wise correspondences between captured images and trajectory data (<xref ref-type="bibr" rid="B142">Stein et al., 2016</xref>). <xref ref-type="fig" rid="F9">Figure 9</xref> displays the process of instance matching and tracking. A counting method for cherry tomatoes based on YOLOv4 was proposed by <xref ref-type="bibr" rid="B165">Wei et al. (2021)</xref>, and it takes the counting problem as detecting and classifying problems that can reduce the effects of occlusion and overlap. <xref ref-type="bibr" rid="B106">Ni et al. (2021)</xref> proposed a method for counting blueberries based on the result of individual 3D berry segmentations. In that study, Mask-R-CNN was used for 2D blueberry detection, and the 3D point was used for 3D reconstruction.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption><p>Instance matching and tracking by 3-D assignment. <bold>(Left)</bold> Key frames extracted from a video sequence with a 1,080-p camera. <bold>(Right)</bold> Graph-based tracking. Each column represents instances found by a neural network, and each color represents an individual grape cluster in a video frame.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g009.tif"/>
</fig>
<p>Some types of fruits are only edible when ripe. Therefore, maturity monition can provide a timely signal to harvest workers. Tomatoes have the characteristics of clustered growth and batch ripening. Immature tomatoes contain solanine, which is noxious to the human body. Thus, dozens of studies are related to tomato maturity detection. <xref ref-type="bibr" rid="B143">Sun et al. (2018)</xref> first used Faster-R-CNN with ResNet 50 to detect critical organs of tomatoes, and the mAP of the model is 0.907. Subsequently, they improved the FPN model to recognize tomato flowers, green tomatoes, and red tomatoes, and the mAP achieved 0.995 (<xref ref-type="bibr" rid="B144">Sun et al., 2020</xref>). Coconuts with different maturities can be sold for various purposes. Therefore, <xref ref-type="bibr" rid="B114">Parvathi and Tamil Selvi (2021)</xref> used Faster-R-CNN to detect the maturities of coconuts in trees to decrease economic loss. The definition of mature and immature fruits is the primary issue of maturity detection. Some researchers transformed the identification task into a classification task. According to the relationship between storage time and appearance, tomatoes can be classified into five categories: &#x201C;Breaker,&#x201D; &#x201C;Turning,&#x201D; &#x201C;Pink,&#x201D; &#x201C;Light red,&#x201D; and &#x201C;Red.&#x201D; A CNN can classify the level of tomato maturity (<xref ref-type="bibr" rid="B191">Zhang L. et al., 2018</xref>). <xref ref-type="bibr" rid="B157">Tu et al. (2018)</xref> collected five maturities category pictures of passion fruit (<xref ref-type="fig" rid="F10">Figure 10</xref>), and then modified the Faster-R-CNN model to recognize the fruit and its ripeness. <xref ref-type="bibr" rid="B152">Tian et al. (2019)</xref> divided objective apples into three classes, young, expanding, and ripe, and optimized the YOLOv3 model with DenseNet for detection. The classification method referred in <xref ref-type="bibr" rid="B152">Tian et al. (2019)</xref> was used on litchi (<xref ref-type="bibr" rid="B161">Wang H. et al., 2021</xref>). However, litchi fruits are different from apples that are small and dense; thus, Wang adjusted the prediction scale and decreased the weight layers of YOLOv3 to enhance the capacity of the model for compact object detection. <xref ref-type="bibr" rid="B78">Khosravi et al. (2021)</xref> coded olives according to their mature stages and varieties, divided them into eight categories, and used a deep convolutional network for detection. The overall accuracy of detection can reach 91.9, and the processing speed on the CPU is 12.64 ms per frame.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption><p>Different maturity levels of passion fruit in <xref ref-type="bibr" rid="B157">Tu et al. (2018)</xref>. <bold>(A)</bold> Near-young passion fruit. <bold>(B)</bold> Young passion fruit. <bold>(C)</bold> Near-mature passion fruit. <bold>(D)</bold> Mature passion fruit. <bold>(E)</bold> After-mature passion fruit.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g010.tif"/>
</fig>
<p>Offering indices of fruit maturity can help workers make harvesting plans and assist harvest robots in making decisions. Some scholars offered indices for describing fruit maturity under the premise of using a CNN to detect fruits. <xref ref-type="bibr" rid="B64">Huang et al. (2020)</xref> utilized Mask-R-CNN to identify the location of tomatoes in images and evaluated the HSV value of detected tomatoes. They then constructed Fuzzy inference rules between the maturity and the color feature of the surface of tomatoes, which can predict ripeness and harvesting schedule. <xref ref-type="bibr" rid="B105">Ni et al. (2020)</xref> also used Mask-R-CNN to extract blueberry fruit traits and gave two indices to describe fruit maturity (<xref ref-type="fig" rid="F11">Figure 11</xref>). One index is about the maturity of individual berries that can infer whether blueberries are harvestable or not. Another is the maturity ratio (mature berry number/total berry number) of a whole cluster that can indicate the specific harvesting time of this cultivar. For clustered and dense fruits such as blueberries, cherries, and cherry tomatoes, the maturity of whole bunches of fruits can be calculated by detecting the maturity of each fruit using DL. At the same time, the labeling process is time-consuming and laborious. To provide technical support for high quality cherry production, <xref ref-type="bibr" rid="B41">Gai et al., 2021</xref> proposed aYOLO-V4-dense model for detection of the maturity of cherries.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption><p>Detection examples in <xref ref-type="bibr" rid="B105">Ni et al. (2020)</xref>. The black rectangle contains the ID number and three traits (number, maturity, and compactness) of the corresponding sample.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g011.tif"/>
</fig>
</sec>
<sec id="S4.SS2.SSS2">
<title>Aerial Platform</title>
<p>Many researchers have begun using UAVs (unmanned aerial vehicles) to obtain images, and UAVs have become common in agricultural remote sensing as intelligent devices progress. Studies have demonstrated that data taken with UAVs are suitable for fruit yield prediction (<xref ref-type="bibr" rid="B166">Wittstruck et al., 2021</xref>). <xref ref-type="bibr" rid="B18">Chen et al. (2017)</xref> proposed a novel method that uses DL to map from input images to total fruit counts. It utilizes a detector based on an FCN model to extract candidate regions in images, and a counting algorithm based on a second convolutional network that estimates the number of fruits in each region. Finally, a linear regression model maps that fruit count estimate to a final fruit count. A UAV-based visual detection technology for green mangoes in trees was proposed by <xref ref-type="bibr" rid="B175">Xiong J. et al. (2020)</xref>. In their study, the YOLOv2 model was trained for green mango identification. The mAP of the trained model on the training set was 86.4%, and estimation error rate was 1.1%. <xref ref-type="bibr" rid="B5">Apolo-Apolo et al. (2020)</xref> used a UAV to monitor citrus in orchards (shown in <xref ref-type="fig" rid="F12">Figure 12</xref>) and adopted Faster-R-CNN to develop a system that can automatically detect and estimate the size of citrus fruits and estimate the total yield of citrus orchards according to detection results. To solve the problem of inconvenient data capture in mountain orchards, <xref ref-type="bibr" rid="B63">Huang et al. (2022)</xref> designed a real-time citrus detection system for yield estimation based on a UAV and the YOLOv5 model. <xref ref-type="bibr" rid="B74">Kalantar et al. (2020)</xref> presented a system for detection and yield estimation of melons with a UAV. The system included three main stages: CNN-based melon recognition, geometric feature extraction (<xref ref-type="bibr" rid="B73">Kalantar et al., 2019</xref>), and individual melon weight (<xref ref-type="bibr" rid="B26">Dashuta and Klapp, 2018</xref>).</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption><p>Workflow of field tests (<xref ref-type="bibr" rid="B5">Apolo-Apolo et al., 2020</xref>).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g012.tif"/>
</fig>
<p>After using UAVs to predict fruit yield produced significant results, some scholars began to use UAVs to detect fruit maturity. <xref ref-type="bibr" rid="B20">Chen et al. (2019)</xref> used a UAV to capture images of the strawberry crop, and then utilized Faster-R-CNN to detect strawberry flowers and immature and mature strawberries with 84.1% accuracy. <xref ref-type="bibr" rid="B199">Zhou et al. (2021)</xref> also divided the growth of strawberries into three stages, &#x201C;flowers,&#x201D; &#x201C;immature fruits,&#x201D; and &#x201C;mature fruits,&#x201D; and utilized the YOLOv3 model to detect images photographed with a UAV. The experimental results show that the model has the best detection effect on the data set taken with the UAV 2 m away from fruits, and the mAP reaches 0.88.</p>
</sec>
<sec id="S4.SS2.SSS3">
<title>Differences Between Two Platforms</title>
<p>In Sections &#x201C;Terrestrial Platform&#x201D; and &#x201C;Aerial Platform,&#x201D; we have described in detail the existing literature on the use of DL for detecting fruits in the growing period, and the differences can be seen in <xref ref-type="table" rid="T5">Table 5</xref>.</p>
<table-wrap position="float" id="T5">
<label>TABLE 5</label>
<caption><p>Summary of related studies on application of CNN-based detection models in growing fruits.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Platform</td>
<td valign="top" align="left">Purpose</td>
<td valign="top" align="left">Detected object<break/> and label</td>
<td valign="top" align="left">CNN-based detection model</td>
<td valign="top" align="left">Following-up works</td>
<td valign="top" align="left">Remarks</td>
<td valign="top" align="left">References</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Terrestrial platform</td>
<td valign="top" align="left">Yield estimation</td>
<td valign="top" align="left">Apple</td>
<td valign="top" align="left">Mask-R-CNN (2D detection)</td>
<td valign="top" align="left">SFM photogrammetry is used for generating 3D point cloud and SVM is used for removing false positive</td>
<td valign="top" align="left">Detection accuracy: 76.2% (2D image detections) and 85.7% (3D detections). Prediction precision: <italic>R</italic><sup>2</sup> = 0.8</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B47">Gen&#x00E9;-Mola et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Apple</td>
<td valign="top" align="left">YOLOv3</td>
<td valign="top" align="left">Counting detected fruits for yield estimation</td>
<td valign="top" align="left">Detection accuracy: 84%</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B67">Janowski et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Mango</td>
<td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="left">The GPS/INS, color cameras with strobes, and LiDAR used for fruit locating, tracking, and counting</td>
<td valign="top" align="left">Prediction accuracy: <italic>R</italic><sup>2</sup> = 0.94</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B142">Stein et al., 2016</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">MangoYOLO</td>
<td valign="top" align="left">Correction factors are used for estimating yield load</td>
<td valign="top" align="left">Detection precision: 98.3%. Estimation precision: 4.6&#x2013;15.2% of packhouse fruit counts</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B80">Koirala et al., 2019a</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Tomato</td>
<td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="left">Stitching detected images and compiling a tomato location map of a greenhouse, estimating tomato size as per bounding box size.</td>
<td valign="top" align="left">Model performance: average precision: 87%, <italic>R</italic><sup>2</sup> = 0.87</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B101">Mu et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Cheery tomato<break/> clusters</td>
<td valign="top" align="left">YOLOv3</td>
<td valign="top" align="left">ResNet-50 is used for classifying fruit clusters and counting total fruit number</td>
<td valign="top" align="left">Prediction precision: RMSE = 6.37, MAPE = 13.9%</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B165">Wei et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Passion fruit</td>
<td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="left">Counting detected fruits for yield estimation</td>
<td valign="top" align="left">Model performance: <italic>P</italic> = 96.2%, <italic>R</italic> = 93.1%, F1 = 0.95</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B156">Tu et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Oliver</td>
<td valign="top" align="left">Inception-ResNetV2</td>
<td valign="top" align="left">Counting detected fruits for yield estimation</td>
<td valign="top" align="left">Model performance: F1 = 0.84</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B6">Aquino et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Grape clusters</td>
<td valign="top" align="left">MobileNet-V2</td>
<td valign="top" align="left">DeepLabV3 segmenting each berry for counting</td>
<td valign="top" align="left">Berry detection accuracy of 94.0% in the VSP and 85.6% in the SMPH</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B187">Zabawa et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Kiwifruit</td>
<td valign="top" align="left">SSD (with MobileNetV2, quantized MobileNetV2, InceptionV3, and quantized InceptionV3)</td>
<td valign="top" align="left">Performing on mobiles with Android system and counting detected fruits for yield estimation</td>
<td valign="top" align="left">True detected rate (TDR) of MobileNetV2, quantized MobileNetV2, InceptionV3, and quantized InceptionV3 are 90.8%, 89.7%, 87.6%, and 72.8%, respectively.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B200">Zhou et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Blueberry</td>
<td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="left">Using different backbones: ResNet101, ResNet50 and MobileNetV1 to Mask-R-CNN and adding a step to outputs each instance of a blueberry to quantify the total number of blueberries in an image.</td>
<td valign="top" align="left">The best result was obtained when the ResNet50 backbone was used achieving a mIoU score of 0.595.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B52">Gonzalez et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Blueberry</td>
<td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="left">3D minimum bounding box calculating fruit cluster compactness after 3D reconstruction and proposing a trait extraction algorithm to segment individual 3D blueberries, count berry number, calculate maturity, and estimate berry size.</td>
<td valign="top" align="left">The average counting accuracy for the 40 samples is 97.3%. The fruit clusters with a low fruit number generally have a higher accuracy, resulting in almost 100% accuracy.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B106">Ni et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Multi-fruit</td>
<td valign="top" align="left">Faster-R-CNN with MIoU</td>
<td valign="top" align="left">Counting detected fruits for yield estimation</td>
<td valign="top" align="left">Model performance: <italic>R</italic><sup>2</sup> of mango, pomegranate, tomato, apple &#x0026; orange are 0.98, 0.92, 0.96, 0.98, and 0.95</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B12">Behera et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Maturity detection</td>
<td valign="top" align="left">Apple (&#x201C;Young Apple,&#x201D; &#x201C;Expanding apple,&#x201D; &#x201C;Ripe apple&#x201D;)</td>
<td valign="top" align="left">YOLOv3</td>
<td valign="top" align="left">Using different data augment methods and data numbers to comparison. Detection under occlusion and overlapping apple conditions and no apple environment.</td>
<td valign="top" align="left">Model performance: F1 = 0.817. Average detection time: 0.304 s</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B152">Tian et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Tomato (&#x201C;Flower,&#x201D; &#x201C;Green tomato,&#x201D; &#x201C;Red tomato&#x201D;)</td>
<td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="left">Taking comparison between YOLOv2, YOLOv3, original Faster-R-CNN, R-FCN, and proposed model.</td>
<td valign="top" align="left">Model performance: Mean average precision: 90.7%. Average test time: 0.073 s. Model memory: 115.9 MB</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B143">Sun et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Tomato (&#x201C;Breakers,&#x201D; &#x201C;Turning,&#x201D; &#x201C;Pink,&#x201D; &#x201C;Light red,&#x201D; &#x201C;Red&#x201D;)</td>
<td valign="top" align="left">Own model</td>
<td valign="top" align="left">Using own designed CNN for images classification</td>
<td valign="top" align="left">Classification accuracy: 91.9%</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B191">Zhang L. et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"></td>
<td valign="top" align="left"/><td valign="top" align="left">Tomato (&#x201C;Immature,&#x201D; &#x201C;Breaker,&#x201D; &#x201C;Preharvest,&#x201D; &#x201C;Harvest&#x201D;)</td>
<td valign="top" align="left">Fuzzing Mask-R-CNN</td>
<td valign="top" align="left">Locating the stalk points of ripe tomatoes by obtaining the contours of tomatoes from Mask-R-CNN for harvesting.</td>
<td valign="top" align="left">Model performance: <italic>P</italic> = 96.1%, <italic>R</italic> = 95.9%.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B64">Huang et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Four blueberry cultivars (&#x201C;Immature&#x201D; and &#x201C;Mature&#x201D;)</td>
<td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="left">Defining and calculating blueberry maturity and compactness. Assessing the extracted traits and delineating trait differences in four blueberry cultivars.</td>
<td valign="top" align="left">Model performance: Mean average precision: 78%. <italic>R</italic><sup>2</sup> of four cultivars: 0.932, 0.877, 0.859, 0.934.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B105">Ni et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Coconut (&#x201C;coconut&#x201D; and &#x201C;Mature coconut&#x201D;)</td>
<td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="left">Comparing the performance of Faster-R-CNN with different backbones, comparing the performance of improved model and other objection detection models.</td>
<td valign="top" align="left">Model performance: Mean average precision: 89.4%. Detection speed: 3.124 s</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B114">Parvathi and Tamil Selvi, 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Passion fruit (&#x201C;After-mature,&#x201D; &#x201C;Mature,&#x201D; &#x201C;Near-mature,&#x201D; &#x201C;Near-young,&#x201D; &#x201C;Young&#x201D;)</td>
<td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="left">Using DSIFT algorithm and LLC algorithm to extract the features of fruit from R, G, B channels and send the representative features to SVM classifier for maturity indentation.</td>
<td valign="top" align="left">Detection accuracy: 92.71% and maturity classification accuracy: 91.52%</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B157">Tu et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Litchi (&#x201C;Ripe litchi,&#x201D; &#x201C;Expanding litchi,&#x201D; &#x201C;Young litchi&#x201D;)</td>
<td valign="top" align="left">YOLOv3-Litchi</td>
<td valign="top" align="left">Comparing the proposed model with YOLOv2, YOLOv3, and Faster-R-CNN.</td>
<td valign="top" align="left">Model performance: average detection time: 0.029 s, mean average precision: 75.6%, average precision of young litchi, expanding litchi, and expanding litchi is 67.3%, 71.9%, 73.8%.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B161">Wang H. et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Oliver (&#x201C;ZIG,&#x201D; &#x201C;RIG,&#x201D; &#x201C;ZVS,&#x201D; &#x201C;RVS,&#x201D; &#x201C;ZBS,&#x201D; &#x201C;RBS,&#x201D; &#x201C;ZOR,&#x201D; &#x201C;ROR&#x201D;)</td>
<td valign="top" align="left">Own model</td>
<td valign="top" align="left">Evaluating the efficiency of six optimizers: Adagrad, SGD, SGDM, RMSProp, Adam, and Nadam.</td>
<td valign="top" align="left">Overall accuracy 91.91%, detection speed: 12.64 ms/frame (CPU)</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B78">Khosravi et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Strawberries (&#x201C;Flower,&#x201D; &#x201C;Flower-Fruit,&#x201D; &#x201C;Green-Fruit,&#x201D; &#x201C;Green-White-Fruit,&#x201D; &#x201C;White-Red-Fruit,&#x201D; &#x201C;Red-Fruit,&#x201D; and &#x201C;Rotted-Fruit&#x201D;)</td>
<td valign="top" align="left">YOLOv3</td>
<td valign="top" align="left">Identify the different ripeness of the detected fruit.</td>
<td valign="top" align="left">The mAP of strawberry maturity classification was 0.89, and the highest classification AP was 0.94 for fully matured fruit.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B186">Yue et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Cherry (&#x201C;Cherry,&#x201D; &#x201C;Cherry_1,&#x201D; &#x201C;Cherry_2&#x201D;)</td>
<td valign="top" align="left">YOLOv4</td>
<td valign="top" align="left">DenseNet is used to replace the CSPDarkNet53 in YOLO-V4 and comparing different models in detecting ripe cherries</td>
<td valign="top" align="left">The mAP increased 0.15 comparing with the YOLO-V4 model and the F1 scores, IOU is 0.947 and 0.856.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B41">Gai et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left">Aerial platform</td>
<td valign="top" align="left">Yield estimation</td>
<td valign="top" align="left">Apple, orange</td>
<td valign="top" align="left">FCN</td>
<td valign="top" align="left">A second neural network and a linear regression were used to count the number of fruit.</td>
<td valign="top" align="left">Mean IU of 0.813 on the oranges and 0.838 on the apples, a best l2 error of 13.8 on the oranges, and 10.5 on the apples</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B18">Chen et al., 2017</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Green mango</td>
<td valign="top" align="left">YOLOv2</td>
<td valign="top" align="left">Counting detected fruits for yield estimation.</td>
<td valign="top" align="left">The mAP was 86.4%, a precision was 96.1% and a recall rate was 89.0%.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B175">Xiong J. et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Citrus</td>
<td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="left">Counting detected fruits and estimate the weight for yield estimation.</td>
<td valign="top" align="left">Mean error is 7.22%.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B5">Apolo-Apolo et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Citrus</td>
<td valign="top" align="left">YOLOv5</td>
<td valign="top" align="left">Comparing the proposed model with different models and different occlusion degrees.</td>
<td valign="top" align="left">Accuracy: 93.32%, speed: 180 ms/frame, FPS: 83 s (In 2080ti), recall: 88.78%</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B63">Huang et al., 2022</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left"/><td valign="top" align="left">Melon</td>
<td valign="top" align="left">RetinaNet</td>
<td valign="top" align="left">Estimate the weight of the detected fruit.</td>
<td valign="top" align="left">Overall average precision score: 0.92 and F1 is more than 0.9</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B74">Kalantar et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Maturity detection</td>
<td valign="top" align="left">Strawberries (&#x201C;Flower,&#x201D; &#x201C;Immature Fruit,&#x201D; &#x201C;Mature Fruit&#x201D;)</td>
<td valign="top" align="left">YOLOv3</td>
<td valign="top" align="left">Identify the different ripeness of the detected fruit.</td>
<td valign="top" align="left">For Flower, Immature Fruit, and Mature Fruit detection from the test data set at 2 m, the APs were 0.83, 0.87, and 0.93, the mAP for the test data set at 2 m was 0.88.</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B199">Zhou et al., 2021</xref></td>
</tr>
</tbody>
</table></table-wrap>
<p>From the above discussion, the advantages and disadvantages of terrestrial and aerial platforms for yield estimation and maturity detection are obvious. For orchards located in harsh terrains, it is time-consuming and laborious that researchers use hand-held cameras to obtain data sets, and it is difficult to achieve automatic detection. Researchers only need to remotely control a UAV to easily acquire a large data set with different terrains and shooting distances, which is more convenient than handheld cameras. However, a UAV cannot be too close to the detected subject in the air; otherwise, a collision accident will occur. Therefore, it is noticed that the operation of a UAV needs more skilled technology.</p>
<p>For the yield prediction task, a UAV can capture a wider field of vision, such as fruits at the top of trees. However, when a UAV is used for long-distance shooting, the visibility of fruits is low because fruits at the bottom or inside of a canopy cannot be recognized, and increase in prediction error. When a handheld camera is used, the visibility of fruits is higher because a small part of a blocked fruit can be detected. However, the repetition rate of photographed fruits is high, which is not conducive to yield estimation.</p>
<p>For the maturity detection task, the characteristics of fruits are more conspicuous when a handheld camera is used for close shooting. Fruits photographed with the UAV equipment are too small because of long distance, and the characteristics are relatively fuzzy. In <xref ref-type="bibr" rid="B199">Zhou et al. (2021)</xref>, researchers used UAV equipment and a handheld camera equipment for data acquisition. They divided the strawberry data captured with the camera into seven different growth stages: flower fruits, green fruits, green-white fruits, white-red fruits, red fruits, and rotten fruits. The strawberry data collected with the UAV were only divided into three labels: flowers, immature fruits, and mature fruits.</p>
</sec>
</sec>
<sec id="S4.SS3">
<title>Fruit Picking</title>
<p>The picking period of fruits arrives when fruit organs expand to a certain size. Mature fruits are needed to harvest fruits in time. However, there has been an imbalance between labor force and economic benefits for a long time. In these years, automatic fruit harvest robots have become a hotspot of intelligent agricultural study. Most fruit trees have proper growth heights and structured planting modes that offer convenience to harvest robots. <xref ref-type="table" rid="T6">Table 6</xref> summarizes the crops (containing fruits, branches, and trunks) experimented on for automatic harvest and corresponding detection models.</p>
<table-wrap position="float" id="T6">
<label>TABLE 6</label>
<caption><p>Summary of related studies on application of CNN-based detection models in fruit harvesting.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Crop applied</td>
<td valign="top" align="left">Basic model</td>
<td valign="top" align="center">Data augment</td>
<td valign="top" align="left">Dataset</td>
<td valign="top" align="center">Transfer learning</td>
<td valign="top" align="center">Detection rate (%)</td>
<td valign="top" align="left">Inference speed (s/image)</td>
<td valign="top" align="left">References</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Apple</td>
<td valign="top" align="left">SSD</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">589 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">89.2</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B117">Peng et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">270 RGB-D images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">86.0</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B189">Zhang J. et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">967 three-modalities images (RGB, range-corrected intensity, and depth)</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">94.8</td>
<td valign="top" align="left">0.074 @548&#x00D7;373 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B45">Gen&#x00E9;-Mola et al., 2019a</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">SSD</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">250 RGB-D images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">92.3</td>
<td valign="top" align="left">2.00 @3840&#x00D7;1080 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B110">Onishi et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">LedNet (FPN+ASPP)</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">1,100 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">85.3</td>
<td valign="top" align="left">0.028 @320&#x00D7;320 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B77">Kang and Chen, 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">12,800 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">87.6</td>
<td valign="top" align="left">0.241 @1920&#x00D7;1080 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B43">Gao et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">800 RGB-D images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">87.1</td>
<td valign="top" align="left">0.124 @1920&#x00D7;1080 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B39">Fu et al., 2020b</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">820 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">92.5</td>
<td valign="top" align="left">0.058 @100&#x00D7;100 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B160">Wan and Goudos, 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">675 RGB-D images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">82.4</td>
<td valign="top" align="left">0.450 @360&#x00D7;640 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B190">Zhang et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">1,140 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">97.3</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B68">Jia et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">24,005 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">58.1</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B30">Dong W. et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">19,528 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">88.0</td>
<td valign="top" align="left">0.250 @1280&#x00D7;720 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B22">Chu et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">DenseNet+FPN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">953 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">93.2</td>
<td valign="top" align="left">0.023 @200&#x00D7;308 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B176">Xu et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left">Citrus</td>
<td valign="top" align="left">SSD</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">1,660 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">91.1</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B117">Peng et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">300 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">85.1</td>
<td valign="top" align="left">0.045 @1024&#x00D7;768 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B94">Liu Y. P. et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">RGB and RGB-HSV images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">97.5</td>
<td valign="top" align="left">0.011 @256&#x00D7;256 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B42">Ganesh et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">5,195 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B172">Xiong et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask-R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">750 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">98.2</td>
<td valign="top" align="left">0.700 @1024&#x00D7;768 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B179">Yang et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">5,195 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">92.2</td>
<td valign="top" align="left">9.230 @1920&#x00D7;1080 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B180">Yang et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">799 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">90.7</td>
<td valign="top" align="left">0.058 @100&#x00D7;100 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B160">Wan and Goudos, 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left">Kiwifruit</td>
<td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">20,160 images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">92.3</td>
<td valign="top" align="left">0.274 @2352&#x00D7;1568 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B36">Fu et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">20,160 images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">87.6</td>
<td valign="top" align="left">0.347 @2352&#x00D7;1568 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B141">Song et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">21,147 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">96.0</td>
<td valign="top" align="left">1.070 @1920&#x00D7;1080 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B100">Mu et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">1,000 NIR images+1,000 RGB images+1,000 RGB-D images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">91.7</td>
<td valign="top" align="left">0.134 @512&#x00D7;424 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B95">Liu et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">YOLOv3</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">20,160 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">90.1</td>
<td valign="top" align="left">0.034 @2352&#x00D7;1568 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B37">Fu et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left">Strawberry</td>
<td valign="top" align="left">SSD</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">4,550 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">87.7</td>
<td valign="top" align="left">0.23 @360&#x00D7;640 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B82">Lamb and Chuah, 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">2,000 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">95.8</td>
<td valign="top" align="left">0.125 @640&#x00D7;480 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B184">Yu et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">81.0</td>
<td valign="top" align="left">0.620 @640&#x00D7;480 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B44">Ge et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask R-CNN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B175">Xiong et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Mask R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">3000 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">78.3</td>
<td valign="top" align="left">0.01 @ 1008&#x00D7;756 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B118">P&#x00E9;rez-Borrero et al., 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">FCN</td>
<td valign="top" align="center"/><td valign="top" align="left">3100 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">93.4</td>
<td valign="top" align="left">0.03 @ 1008&#x00D7;756 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B119">P&#x00E9;rez-Borrero et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left">Grape</td>
<td valign="top" align="left">Mask R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">1,050 RGB-D images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">89.5</td>
<td valign="top" align="left">1.100 @1920&#x00D7; 1080 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B181">Yin et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left">Litchi</td>
<td valign="top" align="left">SSD</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">636 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">86.7</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B117">Peng et al., 2018</xref></td>
</tr>
<tr>
<td valign="top" align="left">Mango</td>
<td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="center"/><td valign="top" align="left">822 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">88.9</td>
<td valign="top" align="left">0.058 @100&#x00D7;100 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B160">Wan and Goudos, 2020</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">DenseNet+FPN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">1694 RGB images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">93.6</td>
<td valign="top" align="left">0.023 @500&#x00D7;500 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B176">Xu et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left"><italic>Rosa roxburghii</italic></td>
<td valign="top" align="left">Faster R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">8,475 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">92.0</td>
<td valign="top" align="left">0.200 @500&#x00D7;500 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B178">Yan et al., 2019</xref></td>
</tr>
<tr>
<td valign="top" align="left">Guava</td>
<td valign="top" align="left">Mask R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">304 RGB-D images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">53.7</td>
<td valign="top" align="left">0.250 @512&#x00D7;424 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B88">Lin et al., 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left">Sweet pepper</td>
<td valign="top" align="left">Faster-R-CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">122 RGB-NIR images</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="center">83.8</td>
<td valign="top" align="left">0.393 @&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B134">Sa et al., 2016</xref></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="left">Deep CNN</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">960 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">82.9</td>
<td valign="top" align="left">&#x2013;</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B127">Rehman and Miura, 2021</xref></td>
</tr>
<tr>
<td valign="top" align="left">Cherry tomato</td>
<td valign="top" align="left">YOLOv3</td>
<td valign="top" align="center">&#x221A;</td>
<td valign="top" align="left">1825 RGB images</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">96.8</td>
<td valign="top" align="left">0.058 @ 1,292&#x00D7;964 px</td>
<td valign="top" align="left"><xref ref-type="bibr" rid="B19">Chen et al., 2021</xref></td>
</tr>
</tbody>
</table></table-wrap>
<sec id="S4.SS3.SSS1">
<title>Fruit Recognition on Fields</title>
<p>The recognition and detection of fruits in an orchard environment provide robots with vital contextual information for maneuvering. However, branches, foliage, and illumination conditions affect the fruit detection with robots. Feature augmentation is a simple way to enhance the learning capacity of DL models. <xref ref-type="bibr" rid="B100">Mu et al. (2019)</xref> collected images with four types of occlusions in four illumination conditions as training data. Some researchers divided target apples into four classes depending on their obscured circumstances: leaf-occluded, branch/wire-occluded, non-occluded, and occluded fruits (<xref ref-type="bibr" rid="B43">Gao et al., 2020</xref>). Different varieties of the same fruit will have subtle differences in appearance. Using Mask-R-CNN to segment fruit images can distinguish fruits from occluded ones well. <xref ref-type="bibr" rid="B22">Chu et al. (2021)</xref> used an integrated data set with two varieties of apple to train Mask-R-CNN for suppression. <xref ref-type="bibr" rid="B68">Jia et al. (2020)</xref> optimized the Mask-R-CNN model in the backbone net, ROI layer, and FCN layer for apple harvesting robots. In a research study on strawberry harvest, the researchers reduced the magnitude of backbone and mask network and used a process of filtering and grouping of candidate regions to replace the object classifier and the bounding box regressor Mask-R-CNN. The new architecture can process original high-resolution images at 10 frames per second (<xref ref-type="bibr" rid="B118">P&#x00E9;rez-Borrero et al., 2020</xref>). Then, <xref ref-type="bibr" rid="B119">P&#x00E9;rez-Borrero et al. (2021)</xref> proposed a new strawberry instance segmentation model based on FCN whose FPS rate was six times higher than those obtained in reference methodologies based on Mask R-CNN.</p>
<p>As we have discussed in Section &#x201C;Dataset Acquisition,&#x201D; a depth image contains more information. <xref ref-type="bibr" rid="B42">Ganesh et al. (2019)</xref> assessed the performance of Mask-R-CNN by applying three forms of color space input, RGB images, HSV images, and RGB + HSV images. The result showed that adding HSV information to RGB images can decrease false positive rate. <xref ref-type="bibr" rid="B134">Sa et al. (2016)</xref> explored two methods for imagery modality fusion based on Faster-R-CNN. One is early fusion (<xref ref-type="fig" rid="F13">Figure 13A</xref>) by augmenting channels of input images from three (red, green and blue) to four (red, green, blue, and NIR) channels. Another is later fusion (<xref ref-type="fig" rid="F13">Figure 13B</xref>) that fuses pieces of classified information of an RGB-trained model and an NIR-trained model. NIR (near infrared) here refers to images taken by near-infrared imaging technology. There are also two fusion methods for detecting kiwifruits based on Faster-R-CNN (<xref ref-type="bibr" rid="B95">Liu et al., 2019</xref>). One is similar to the early fusion (<xref ref-type="bibr" rid="B134">Sa et al., 2016</xref>), and the other fuses the feature maps from two modes displayed in <xref ref-type="fig" rid="F14">Figure 14</xref>. The background objects of RGB-D images captured with a Kinect V2 camera can be filtered by distance threshold and foreground-RGB images, and Faster-R-CNN with VGG achieved a high average precision of 0.893 for the foreground-RGB-images (<xref ref-type="bibr" rid="B39">Fu et al., 2020b</xref>). <xref ref-type="bibr" rid="B46">Gen&#x00E9;-Mola et al. (2019b)</xref> added an imaging modality, the range-corrected IR intensity proportional to reflectance, based on RGB-D images. It makes an input image become five channels, and the F1-score of the detection model improves 4.46% more than simple RGB images.</p>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption><p>Diagram of fusion methods in <xref ref-type="bibr" rid="B134">Sa et al. (2016)</xref>. <bold>(A)</bold> Early fusion: first, channels of the detected image are augmented from three to four channels. Second, the augmented image is detected by Faster-R-CNN. Third, NMS (non-maximum suppression) removes duplicate predictions. Finally, the classifier and regressor calculate the category and coordinate of the bounding box. <bold>(B)</bold> Late fusion: first, the RGB image and the NIR image are detected by Faster-R-CNN. Second, the detected outputs from two Faster R-CNN networks are fused. Third, NMS (non-maximum suppression) removes duplicate predictions. Finally, the classifier and regressor calculate the category and coordinate of the bounding box.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g013.tif"/>
</fig>
<fig id="F14" position="float">
<label>FIGURE 14</label>
<caption><p>Feature-fusion model in <xref ref-type="bibr" rid="B95">Liu et al. (2019)</xref>. First, it inputs the RGB and NIR images separately into two VGG16 networks and then combined them on the feature map; then, the feature map is detected by Faster-R-CNN.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g014.tif"/>
</fig>
<p>In most studies, researchers spent energy optimizing algorithms. <xref ref-type="bibr" rid="B117">Peng et al. (2018)</xref> used SDD and replaced the original VGG-16 with ResNet-101 to detect apple, citrus, and lichi. Besides, decreasing layers of the backbone of SSD can achieve accurate and precise detection in a low-power hardware (<xref ref-type="bibr" rid="B82">Lamb and Chuah, 2018</xref>). <xref ref-type="bibr" rid="B77">Kang and Chen (2020)</xref> designed a CNN model named &#x201C;LedNet,&#x201D; which is mainly improved by a lightweight backbone, FPN, and ASSP, for fruit detection in an apple orchard. Integration of DenseNet and FPN can obtain small fruits&#x2019; features more correctly (<xref ref-type="bibr" rid="B176">Xu et al., 2021</xref>). <xref ref-type="bibr" rid="B36">Fu et al. (2018)</xref> first used a DL model for kiwifruit detection in 2018, and they developed a kiwifruit detection system based on Faster-R-CNN with ZFNet for filed images. Three years later, they proposed a DY3TNet model based on the addition of convolutional layers to YOLOv3-Tiny for kiwifruit recognition in a wild environment (<xref ref-type="bibr" rid="B37">Fu et al., 2021</xref>). Some scholars are also dedicated to kiwifruit detection but used Faster-R-CNN with VGG-16; however, the precision and speed of detection are lower than the results of <xref ref-type="bibr" rid="B36">Fu et al. (2018)</xref>. Modification of the pooling layer can also improve detection accuracy. <xref ref-type="bibr" rid="B178">Yan et al. (2019)</xref> changed the Faster-R-CNN model by replacing the ROI pooling layer with the ROI align layer. <xref ref-type="bibr" rid="B160">Wan and Goudos (2020)</xref> modified the pooling layers and convolution layers of the existing Faster-R-CNN. In the two experiments (<xref ref-type="bibr" rid="B178">Yan et al., 2019</xref>; <xref ref-type="bibr" rid="B160">Wan and Goudos, 2020</xref>), detection speed and accuracy accomplished prominent improvements. As we know, most fruits are elliptical in a 2D space. Thus, specialists presented an ellipse regression model based on Mask-R-CNN for detecting elliptical objects and inferring occluded elliptical objects (<xref ref-type="bibr" rid="B30">Dong W. et al., 2021</xref>). The original YOLOv3 has low precision in detecting cherry tomatoes, and DPNs (dual-path networks) can extract richer features of recognition targets. Therefore, researchers improved the YOLOv3 model based on DPNs for identification of cherry tomatoes.</p>
</sec>
<sec id="S4.SS3.SSS2">
<title>Obstacle Avoidance</title>
<p>Robots should also learn to avoid foliage and branches except when identifying fruits. For sure, researchers thought of making robots recognize obstructions while detecting fruits, so robots can react differently according to different objects. Using the R-CNN model to detect and locate branches of apple trees in natural environments can establish a branch of skeletons, so that the arms of robots can avoid branches while grabbing apples (<xref ref-type="bibr" rid="B189">Zhang J. et al., 2018</xref>). For citrus harvest, <xref ref-type="bibr" rid="B179">Yang et al. (2019)</xref> utilized the Mask-R-CNN model to recognize and reconstruct branches of citrus trees. Later, the researchers designed a recognition model based on their previous studies for citrus harvest robots to detect fruits and branches simultaneously (<xref ref-type="bibr" rid="B180">Yang et al., 2020</xref>). <xref ref-type="bibr" rid="B88">Lin et al. (2021)</xref> used a tiny Mask-R-CNN model to identify fruits and branches of guava trees and reconstructed the fruits and branches for robotic harvest.</p>
<p>There are some other means for occlusion avoidance except when detecting obstructions. <xref ref-type="bibr" rid="B127">Rehman and Miura (2021)</xref> presented a viewpoint plan for fruit harvest. They demonstrated the possible types of a fruit in one scene with the labels &#x201C;center,&#x201D; &#x201C;left,&#x201D; &#x201C;right,&#x201D; &#x201C;occluded,&#x201D; which are depicted in <xref ref-type="fig" rid="F15">Figure 15</xref>. The arm of a robot is qualified to determine the harvesting path as per detected labels. What is more, objective fruits could be classified into normal, branch occlusion leaf occlusion, slight occlusion overlapping, or main branch (<xref ref-type="bibr" rid="B94">Liu Y. P. et al., 2018</xref>). Also, a new strawberry-harvesting robot with a more sophisticated active obstacle separation strategy has been developed, and the strawberry location detector in the system is based on Mask-R-CNN (<xref ref-type="bibr" rid="B175">Xiong et al., 2020</xref>).</p>
<fig id="F15" position="float">
<label>FIGURE 15</label>
<caption><p>Possible types of fruit in one scene formulated by <xref ref-type="bibr" rid="B127">Rehman and Miura (2021)</xref>. <bold>(A)</bold> Center, <bold>(B)</bold> left, <bold>(C)</bold> right, and <bold>(D)</bold> occluded.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g015.tif"/>
</fig>
</sec>
<sec id="S4.SS3.SSS3">
<title>Picking Point Detection</title>
<p>The feasibility of automatic harvesting has been confirmed broadly. A further important issue is locating harvesting points precisely that can guarantee that the robot&#x2019;s grasp of fruits is accurate and uninjurious. Mask-R-CNN not only can detect an object accurately but can also generate corresponding masks of an object region at the pixel level, which can assist in locating picking points. <xref ref-type="bibr" rid="B96">Longye et al. (2019)</xref> segmented and reconstructed the overlapping citrus using the Mask-R-CNN model and performing concave region simplification and distance analysis. Strawberry detection can also employ the Mask-R-CNN model. Then, picking points are determined by analyzing the shape and edge of objective masks (<xref ref-type="bibr" rid="B183">Yu et al., 2018</xref>). <xref ref-type="bibr" rid="B44">Ge et al. (2019)</xref> also utilized the Mask-R-CNN model to detect strawberries based on RGB-D images that have depth information of images; they performed coordinate transformation and density-based point clustering, and proposed a location approximation method to help robots locate strawberry fruits. <xref ref-type="bibr" rid="B181">Yin et al. (2021)</xref> proposed segmenting the contours of grapes from RGB images with Mask-R-CNN and then reconstructing a grape model by fitting a cylinder model based on point cloud data extracted from segmented images. By recognizing and calculating the outline of a bunch of grapes, the arm of robot can grab stalks at the top of a bunch of grapes. Shake-and-catch harvesting first appeared in 2010 (<xref ref-type="bibr" rid="B58">He L. et al., 2017</xref>). Some researchers used the Faster-R-CNN model to establish a relationship between fruit location and branch location (<xref ref-type="bibr" rid="B190">Zhang et al., 2020</xref>). Connections can help a robot to determine shake points.</p>
<p>Generally, researchers detect fruits on the side of trees, but <xref ref-type="bibr" rid="B110">Onishi et al. (2019)</xref> proposed a novel method for inspecting apples from below. The SSD model is used to detect the 2-D position of the apple shown in <xref ref-type="fig" rid="F16">Figure 16A</xref>. The stereo camera ZED provides the 3-D position of the center of the bounding box, which is like in <xref ref-type="fig" rid="F16">Figure 16B</xref>, and the position can be a picking point. Then, the robot can move below the target apple to grasp the fruit according to the predicted position like in <xref ref-type="fig" rid="F16">Figure 16C</xref>.</p>
<fig id="F16" position="float">
<label>FIGURE 16</label>
<caption><p>Automatic apple harvesting mode in <xref ref-type="bibr" rid="B110">Onishi et al. (2019)</xref>. <bold>(A)</bold> Detection of a two-dimensional position, <bold>(B)</bold> detection of a three-dimensional position, <bold>(C)</bold> approaching the target apple.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g016.tif"/>
</fig>
</sec>
</sec>
<sec id="S4.SS4">
<title>Fruit Grading</title>
<p>After a fruit is picked, it will gradually flow to the market and produce economic benefits. Recently, customers have higher requirements for fruit quality as consumption levels increase. Hence, it is necessary to evaluate the quality of fruits before delivering them to consumers because of external and internal vulnerabilities. Those with better fructifications can be consumed, and those with worse can be processed to make fruit foods. Graded-based vendition by detecting internal diseases, sugar content, surface damages, maturity, size, etc. can promise both seller and purchaser benefits. In this section, we will introduce the research on CNN-based fresh fruit grading from grading as per external traits, grading as per internal traits, and fruit cultivar classification.</p>
<sec id="S4.SS4.SSS1">
<title>External Trait-Based Grading</title>
<p>External phenotypic characteristics of fruits directly show their qualities, which affect the sale price and consumer enthusiasm. Thus, external quality detection plays a significant role in fruit grading. Many experiments testified that CNNs have noteworthy superiority in fruit quality grading (<xref ref-type="bibr" rid="B163">Wang et al., 2018</xref>; <xref ref-type="bibr" rid="B66">Jahanbakhshi et al., 2020</xref>; <xref ref-type="bibr" rid="B116">Patil et al., 2021</xref>). In the research of <xref ref-type="bibr" rid="B163">Wang et al. (2018)</xref>, a modified AlexNet model was used to extract the feature of defects on litchi surface and classify litchi defect images. The classification precision of the AlexNet-based full convolutional network is higher than that of linear SVM and Naive Bayes Classifier. <xref ref-type="bibr" rid="B66">Jahanbakhshi et al. (2020)</xref> compared sour lemon detection performance based on a CNN model with other image categorization methods and demonstrated the superiority of the CNN-based model in fruit grading. <xref ref-type="bibr" rid="B116">Patil et al. (2021)</xref> also concluded that CNNs have a faster speed of operation in dragon fruit grading and sorting by comparing the performance of ANN, s, and CNN models.</p>
<p>Apple is the most salable and lucrative fruit globally. Some researchers developed apple defect detection systems for apple grading. <xref ref-type="bibr" rid="B32">Fan et al. (2020)</xref> designed a 4-lane fruit sorting system to detect and sort defective apples, and a CNN model for a defective apple sorting system, in which a global average pooling layer was applied to replace a fully connected layer. Wu, Zhu, and Ren performed laser-induced light backscattering imaging to capture apple defect images and designed a simple CNN model to classify scabs on apple surface (<xref ref-type="bibr" rid="B168">Wu A. et al., 2020</xref>). Aside from scabs on apple surface, the CNN model can classify images of apples with bruises, cracks, and cuts (<xref ref-type="bibr" rid="B107">Nur Alam et al., 2020</xref>). Researchers also conducted related studies on other fruits. <xref ref-type="bibr" rid="B8">Azizah et al. (2017)</xref> used a CNN model to implement mangosteen surface defect detection. <xref ref-type="bibr" rid="B188">Zeng et al. (2019)</xref> constructed an ensemble-convolution neural net (E-CNN) model based on the &#x201C;Bagging&#x201D; learning method for detection of defects in jujube fruits. Cherries are prone to abnormal shapes during growth, so some researchers used a modified AlexNet model to classify cherries according to growth shapes (<xref ref-type="bibr" rid="B99">Momeny et al., 2020</xref>). <xref ref-type="bibr" rid="B170">Wu S. et al. (2020)</xref> combined and investigated several deep learning methods for detecting visible mango defects and found that VGG-16 has a dominant position by combining and investigating several DL methods. <xref ref-type="bibr" rid="B27">De Luna et al. (2019)</xref> also demonstrated that the VGG-16 model has better performance in tomato defect inspection. Some researchers used a modified ResNet-50 model to extract the features of tomato surface defects and classify images of tomato defects (<xref ref-type="bibr" rid="B25">Da Costa et al., 2020</xref>). <xref ref-type="bibr" rid="B19">Chen et al. (2021)</xref> established an online citrus sorting system, shown in <xref ref-type="fig" rid="F17">Figure 17</xref>, and a detector named Mobile-citrus based on Mobile-V2 to identify surface defects in citrus. Then, the arms of robots arms pick out the defective ones with the linear Kalman filter model used in predicting the future path of the fruits.</p>
<fig id="F17" position="float">
<label>FIGURE 17</label>
<caption><p>Platform setup and computer vision system (<xref ref-type="bibr" rid="B19">Chen et al., 2021</xref>). <bold>(A)</bold> The citrus processing line was assembled in the laboratory, with a webcam mounted above the conveyor. <bold>(B)</bold> The diagram shows an automated citrus sorting system using a camera and robot arms, and the robot arms will be implemented in future studies.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-868745-g017.tif"/>
</fig>
<p>The external appearance of a fruit sometimes also represents its freshness. A multi-class classifier based on VGG-16 and Inception-V3 was built by <xref ref-type="bibr" rid="B7">Ashraf et al. (2019)</xref> for detecting fresh and rotten fruits. Researchers also practiced the advantages of CNNs in classifying the freshness of apples, bananas, and oranges (<xref ref-type="bibr" rid="B3">Ananthanarayana et al., 2020</xref>).</p>
</sec>
<sec id="S4.SS4.SSS2">
<title>Internal Trait-Based Grading</title>
<p>Commonly used RGB images cannot acquire internal traits of fruits, for instance, diseases, sugar content, moisture, etc. Consequently, many researchers combined CNN-based DL models with spectrum techniques and made remarkable progress in internal quality-based grading. The sweetness, crispiness, and moisture of apples can be detected using hyperspectral images and 3D-CNN (<xref ref-type="bibr" rid="B162">Wang et al., 2020</xref>). Researchers have also proposed a multi-task model based on 3D-CNN for predicting the sugar content and hardness of yellow peaches simultaneously (<xref ref-type="bibr" rid="B177">Xu et al., 2020</xref>). <xref ref-type="bibr" rid="B71">Jie et al. (2021)</xref> proposed a non-destructive determination method based on the YOLOv3 algorithm, and hyperspectral imaging technology contraposes citrus granulation.</p>
</sec>
</sec>
</sec>
<sec id="S5">
<title>Challenges and Future Perspective</title>
<p>As per the above statements, the appearance of CNN models is already invigorating the automatic production of fresh fruits. However, people remain having quite a lot of challenges to face, because the whole automation of the fruit industry is merely in the period of development.</p>
<sec id="S5.SS1">
<title>Environmental Issues</title>
<p>The problem of fruits being occluded is a difficulty in fruit detection. Most occlusions are caused by foliage, branches, trunks, and fruit overlapping in complex fruit-growing environments. Moreover, varying illumination conditions are also one of the instability factors in fruit detection. For instance, green fruits, such as green citrus, green litchi, avocado, and guava, conceal in a green background, which results in more faulty detections of machine visions. Thus, algorithms with high detection accuracy and speed are the objective of researchers.</p>
<p>In addition to algorithm improvement, human intervention can also assist in solving environmental issues. It is a feasible method to increase the visibility of fruits by trimming the crown of fruit trees and standardizing planting according to the principles of horticultural operations. For example, a trellised fruiting wall is suitable for robotic operations during pruning and harvesting (<xref ref-type="bibr" rid="B97">Majeed et al., 2020</xref>). Artificially improving the lighting of an environment can also reduce uncertainty in the process of detection. When light is strong, cameras are prone to overexposure. In response to this problem, some researchers have adopted a shading platform to reduce the impact of sun exposure (<xref ref-type="bibr" rid="B51">Gongal et al., 2016</xref>; <xref ref-type="bibr" rid="B104">Nguyen et al., 2016</xref>; <xref ref-type="bibr" rid="B139">Silwal et al., 2016</xref>). To increase the utilization rate of machines, people will have to let robots work at night. However, there is insufficient lighting during night operations, and external light sources are needed to improve the lighting of an environment (<xref ref-type="bibr" rid="B80">Koirala et al., 2019a</xref>). Most of the current shading devices and light supply devices are relatively bulky, so it is of commercial value to design a shading or a lighting system that is simpler and more portable.</p>
</sec>
<sec id="S5.SS2">
<title>Exploration of New Areas</title>
<p>In the process of fresh fruit production from blooming to marketing, and pollination, pesticide application, harvesting, sorting, and grading all need a large pool of workers. The preceding discussion suggests that most applications of CNNs in fresh fruit production are in the algorithm development stage. Autonomous operation of robots is mostly used for fruit harvesting and grading. There are fewer exploitations of automatic pollination robots for the problem of greenhouse plants&#x2019; insufficient pollination. In current studies, Chunjiang Zhao utilized the improved YOLOv3 network to identify tomato flowers in greenhouses and embedded the system in automatic pollination robots. Phenology distribution monitoring can govern the timing and dosage of chemistry thinning, which determines the quality of fruits. Fruit flower phenology involves a period from the emergence of fruit buds to petal withering means that monitoring of flower phenology is not only estimating flower number. Studies on using computer vision to detect fruit flower phenology are rare, and CNN-based methods are even less. According to our search, <xref ref-type="bibr" rid="B164">Wang X. et al. (2021)</xref> designed a phenology detection model based on a CNN named DeepPhenology to estimate apple flower phenology distribution. Currently, more researchers are utilizing CNN to detect fruit flowers and achieve the purpose of yield estimation. Perhaps the application of CNN in fruit flowers phenology estimation is a new area worth exploring.</p>
<p>Food safety is an issue that concerns people, because accumulation of pesticides in the human body risks causing cancers. However, pesticide residues on fruit surfaces are inescapable, because orchardists will perform pesticide delivery to guarantee fruit&#x2019;s healthy growth. CNNs can be used to identify pesticide residues, but the CNN used in most studies (<xref ref-type="bibr" rid="B182">Yu et al., 2021</xref>; <xref ref-type="bibr" rid="B202">Zhu et al., 2021</xref>) is a one-dimensional CNN, and input data are pre-processing data extracted with a spectrometer. The process of detection is complicated and cumbersome. Rarely have researchers used the 2D CNN model to detect pesticide residues in harvested fruits (<xref ref-type="bibr" rid="B70">Jiang et al., 2019</xref>). Although pesticide residues belong to the external characteristics of fruits, its vision detection still needs hyperspectral images, because RGB images cannot capture pesticide residues. The current detection methods have complex processes out of proportion to the economic benefits generated by pesticide residue detection. Thus, the feasibility of using CNNs to detect pesticide residues in fruits should be studied further. When grading and sorting clustered fruits such as grapes, litchis, and longan, a manipulator grabs the stalk on the top of a fruit to minimize damage to the fruit. However, fruits on the sorting table are arranged disorderly, and stalks are not arranged neatly on a horizontal plane. Therefore, it is necessary to use CNNs to determine the robot&#x2019;s sequence of grabbing of clustered fruits (<xref ref-type="bibr" rid="B192">Zhang and Gao, 2020</xref>).</p>
<p>There is no doubt that CNNs have a developing potential in fresh fruit production. In future studies, it is promising to enhance the application areas of CNNs in fresh fruit detection. It could be a good direction that infuses CNNs into whole fruit production.</p>
</sec>
<sec id="S5.SS3">
<title>Execution of Multiple Tasks</title>
<p>Fruit surfaces are easily damaged, so the general method is utilizing a mechanical arm to grab fruits to reduce mechanical injuries. Most existing CNN-based picking robots are based on one fruit kind, However, the time of fruit harvest is not continuous, therefore, robots are, most, of the time idle. That generates averse economic effectiveness, because robots have high manufacturing expenses but low use ratio. According to the advantages of CNNs, they can directly extract features from input images; therefore, scholars can develop algorithms that can detect and locate a variety of fruits (<xref ref-type="bibr" rid="B135">Saedi and Khosravi, 2020</xref>). The mode of multitask operations can improve the use ratio of harvest robots that ensures fruit harvest robots&#x2019; commercial value.</p>
<p>In CNN-based fruit quality grading, detection methods based on RGB images can only identify external defects, and detection methods based on hyperspectral and infrared images are focused more on internal trait detection. Results of a single detection technique are biased. Simultaneous detection of multiple quality parameters and comprehensive evaluation are a good improving trend. In addition, detection algorithms and hardware should be optimized with increasing detection difficulty.</p>
</sec>
</sec>
<sec id="S6" sec-type="conclusion">
<title>Conclusion</title>
<p>The perishability and fragility of fruits make fruits use more labor force for careful care during the production process, which is also the reason why most fruits are expensive. At present, many researchers are bringing artificial intelligence into the field of fruit production and are carrying out a series of research studies on the use of machine vision to identify fruits. In this article, the principle of CNNs and implementation of CNN-based detection methods is elaborated, enabling researchers to better understand CNNs and their applications in fruit detection. This review emphasizes the application of CNNs in fresh fruit production, including detection of fruit flowers, detection of fruits in the expansion period, detection of fruits in the harvest period, and detection of fruits before entering the market. We have performed a lot of investigations and analyses of literature in this area and presented in detail the convolution models, improvement points, training methods, detected objects, and final detection results in these studies. Through our investigation of experiments, we found that CNNs do have exceptional performance in the detection of fruits. However, this does not mean that fruit detection should evolve toward a single direction of detection based on CNNs. Through our comprehension and comparison of current research, we summarized the challenges that researchers encountered when using CNNs for fruit recognition and discussed future development trends.</p>
</sec>
<sec id="S7">
<title>Author Contributions</title>
<p>CW, JX, and ZZ designed the survey. SL, BZ, LL, GL, YW, and PH collected and analyzed the data, and wrote the manuscript. JX and ZZ revised the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S8" sec-type="funding-information">
<title>Funding</title>
<p>This research was supported by the National Natural Science Foundation of China (52005069 and 32071912) and the China Postdoctoral Science Foundation (2020M683379).</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abadi</surname> <given-names>M.</given-names></name> <name><surname>Agarwal</surname> <given-names>A.</given-names></name> <name><surname>Barham</surname> <given-names>P.</given-names></name> <name><surname>Brevdo</surname> <given-names>E.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Citro</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>TensorFlow: large-scale machine learning on heterogeneous distributed systems.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. Available online at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1603.04467">https://arxiv.org/abs/1603.04467</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alzubaidi</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Humaidi</surname> <given-names>A. J.</given-names></name> <name><surname>Al-Dujaili</surname> <given-names>A.</given-names></name> <name><surname>Duan</surname> <given-names>Y.</given-names></name> <name><surname>Al-Shamma</surname> <given-names>O.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Review of deep learning: concepts, CNN architectures, challenges, applications, future directions.</article-title> <source><italic>J. Big Data</italic></source> <volume>8</volume>:<issue>53</issue>. <pub-id pub-id-type="doi">10.1186/s40537-021-00444-8</pub-id> <pub-id pub-id-type="pmid">33816053</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ananthanarayana</surname> <given-names>T.</given-names></name> <name><surname>Ptucha</surname> <given-names>R.</given-names></name> <name><surname>Kelly</surname> <given-names>S. C.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep learning based fruit freshness classification and detection with CMOS image sensors and edge processors.</article-title> <source><italic>Electron. Imaging</italic></source> <volume>2020</volume> <fpage>172-1</fpage>&#x2013;<lpage>172-7</lpage>.</citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Angermueller</surname> <given-names>C.</given-names></name> <name><surname>P&#x00E4;rnamaa</surname> <given-names>T.</given-names></name> <name><surname>Parts</surname> <given-names>L.</given-names></name> <name><surname>Stegle</surname> <given-names>O.</given-names></name></person-group> (<year>2016</year>). <article-title>Deep learning for computational biology.</article-title> <source><italic>Mol. Syst. Biol.</italic></source> <volume>12</volume>:<issue>878</issue>. <pub-id pub-id-type="doi">10.15252/msb.20156651</pub-id> <pub-id pub-id-type="pmid">27474269</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apolo-Apolo</surname> <given-names>O. E.</given-names></name> <name><surname>Mart&#x00ED;nez-Guanter</surname> <given-names>J.</given-names></name> <name><surname>Egea</surname> <given-names>G.</given-names></name> <name><surname>Raja</surname> <given-names>P.</given-names></name> <name><surname>P&#x00E9;rez-Ruiz</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep learning techniques for estimation of the yield and size of citrus fruits using a UAV.</article-title> <source><italic>Eur. J. Agron.</italic></source> <volume>115</volume>:<issue>126030</issue>. <pub-id pub-id-type="doi">10.1016/j.eja.2020.126030</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aquino</surname> <given-names>A.</given-names></name> <name><surname>Ponce</surname> <given-names>J. M.</given-names></name> <name><surname>And&#x00FA;jar</surname> <given-names>J. M.</given-names></name></person-group> (<year>2020</year>). <article-title>Identification of olive fruit, in intensive olive orchards, by means of its morphological structure using convolutional neural networks.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>176</volume>:<issue>105616</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105616</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ashraf</surname> <given-names>S.</given-names></name> <name><surname>Kadery</surname> <given-names>I.</given-names></name> <name><surname>Chowdhury</surname> <given-names>A. A.</given-names></name> <name><surname>Mahbub</surname> <given-names>T. Z.</given-names></name> <name><surname>Rahman</surname> <given-names>R. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Fruit image classification using convolutional neural networks.</article-title> <source><italic>Int. J. Softw. Innov.</italic></source> <volume>7</volume> <fpage>51</fpage>&#x2013;<lpage>70</lpage>. <pub-id pub-id-type="doi">10.4018/IJSI.2019100103</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Azizah</surname> <given-names>L. M.</given-names></name> <name><surname>Umayah</surname> <given-names>S. F.</given-names></name> <name><surname>Riyadi</surname> <given-names>S.</given-names></name> <name><surname>Damarjati</surname> <given-names>C.</given-names></name> <name><surname>Utama</surname> <given-names>N. A.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Deep learning implementation using convolutional neural network in mangosteen surface defect detection</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 7th IEEE International Conference on Control System, Computing and Engineering (ICCSCE)</italic></source>, <publisher-loc>Penang</publisher-loc>.</citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Badrinarayanan</surname> <given-names>V.</given-names></name> <name><surname>Kendall</surname> <given-names>A.</given-names></name> <name><surname>Cipolla</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>SegNet: a deep convolutional encoder-decoder architecture for image segmentation.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>39</volume> <fpage>2481</fpage>&#x2013;<lpage>2495</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2016.2644615</pub-id> <pub-id pub-id-type="pmid">28060704</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bao</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>D.</given-names></name> <name><surname>Wen</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Hua</surname> <given-names>G.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>CVAE-GAN: fine-grained image generation through asymmetric training</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 7th IEEE International Conference on Control System, Computing and Engineering (ICCSCE)</italic>,</source> <publisher-loc>Penang</publisher-loc>.</citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bargoti</surname> <given-names>S.</given-names></name> <name><surname>Underwood</surname> <given-names>J. P.</given-names></name></person-group> (<year>2017</year>). <article-title>Image segmentation for fruit detection and yield estimation in apple orchards.</article-title> <source><italic>J. Field Robot.</italic></source> <volume>34</volume> <fpage>1039</fpage>&#x2013;<lpage>1060</lpage>. <pub-id pub-id-type="doi">10.1002/rob.21699</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Behera</surname> <given-names>S. K.</given-names></name> <name><surname>Rath</surname> <given-names>A. K.</given-names></name> <name><surname>Sethy</surname> <given-names>P. K.</given-names></name></person-group> (<year>2021</year>). <article-title>Fruits yield estimation using faster R-CNN with MIoU.</article-title> <source><italic>Multimed. Tools Appl.</italic></source> <volume>80</volume> <fpage>19043</fpage>&#x2013;<lpage>19056</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-021-10704-7</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bochkovskiy</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Liao</surname> <given-names>H. M.</given-names></name></person-group> (<year>2020</year>). <article-title>YOLOv4: optimal speed and accuracy of object detection.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. Available online at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/pdf/2004.10934.pdf">https://arxiv.org/pdf/2004.10934.pdf</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bolya</surname> <given-names>D.</given-names></name> <name><surname>Zhou</surname> <given-names>C.</given-names></name> <name><surname>Xiao</surname> <given-names>F.</given-names></name> <name><surname>Lee</surname> <given-names>Y. J.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>YOLACT: real-time instance segmentation</article-title>,&#x201D; in <source><italic>Proceedings of the 2019 IEEE/CVF International Conference on Computer Vision (ICCV)9156-9165</italic></source>, <publisher-loc>Seoul</publisher-loc>. <pub-id pub-id-type="doi">10.1109/ICCV.2019.00925</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bulanon</surname> <given-names>D. M.</given-names></name> <name><surname>Burks</surname> <given-names>T. F.</given-names></name> <name><surname>Alchanatis</surname> <given-names>V.</given-names></name></person-group> (<year>2009</year>). <article-title>Image fusion of visible and thermal images for fruit detection.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>103</volume> <fpage>12</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2009.02.009</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chawla</surname> <given-names>N. V.</given-names></name> <name><surname>Bowyer</surname> <given-names>K. W.</given-names></name> <name><surname>Hall</surname> <given-names>L. O.</given-names></name> <name><surname>Kegelmeyer</surname> <given-names>W. P.</given-names></name></person-group> (<year>2002</year>). <article-title>SMOTE: synthetic minority over-sampling technique.</article-title> <source><italic>J. Artif. Intell. Res.</italic></source> <volume>16</volume> <fpage>321</fpage>&#x2013;<lpage>357</lpage>. <pub-id pub-id-type="doi">10.1613/jair.953</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Papandreou</surname> <given-names>G.</given-names></name> <name><surname>Kokkinos</surname> <given-names>I.</given-names></name> <name><surname>Murphy</surname> <given-names>K.</given-names></name> <name><surname>Yuille</surname> <given-names>A. L.</given-names></name></person-group> (<year>2018</year>). <article-title>DeepLab: semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected CRFs.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>40</volume> <fpage>834</fpage>&#x2013;<lpage>848</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2017.2699184</pub-id> <pub-id pub-id-type="pmid">28463186</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S. W.</given-names></name> <name><surname>Shivakumar</surname> <given-names>S. S.</given-names></name> <name><surname>Dcunha</surname> <given-names>S.</given-names></name> <name><surname>Das</surname> <given-names>J.</given-names></name> <name><surname>Okon</surname> <given-names>E.</given-names></name> <name><surname>Qu</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Counting apples and oranges with deep learning: a data-driven approach.</article-title> <source><italic>IEEE Robot. Autom. Lett.</italic></source> <volume>2</volume> <fpage>781</fpage>&#x2013;<lpage>788</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2017.2651944</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>An</surname> <given-names>X.</given-names></name> <name><surname>Gao</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Kang</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>A deep learning-based vision system combining detection and tracking for fast on-line citrus sorting.</article-title> <source><italic>Front. Plant Sci.</italic></source> <volume>12</volume>:<issue>622062</issue>. <pub-id pub-id-type="doi">10.3389/fpls.2021.622062</pub-id> <pub-id pub-id-type="pmid">33643351</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Lee</surname> <given-names>W. S.</given-names></name> <name><surname>Gan</surname> <given-names>H.</given-names></name> <name><surname>Peres</surname> <given-names>N.</given-names></name> <name><surname>Fraisse</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Strawberry yield prediction based on a deep neural network using high-resolution aerial orthoimages.</article-title> <source><italic>Remote Sens.</italic></source> <volume>11</volume>:<issue>1584</issue>. <pub-id pub-id-type="doi">10.3390/rs11131584</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>Boundary-preserving mask R-CNN</article-title>,&#x201D; in <source><italic>Proceedings of the European Conference on Computer Vision</italic></source>, <publisher-loc>Glasgow</publisher-loc>, <fpage>660</fpage>&#x2013;<lpage>676</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-58568-6_39</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Lammers</surname> <given-names>K.</given-names></name> <name><surname>Lu</surname> <given-names>R.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning-based apple detection using a suppression mask R-CNN.</article-title> <source><italic>Pattern Recogn. Lett.</italic></source> <volume>147</volume> <fpage>206</fpage>&#x2013;<lpage>211</lpage>. <pub-id pub-id-type="doi">10.1016/j.patrec.2021.04.022</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cortes</surname> <given-names>C.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>1995</year>). <article-title>Support-vector networks.</article-title> <source><italic>Mach. Learn.</italic></source> <volume>20</volume> <fpage>273</fpage>&#x2013;<lpage>297</lpage>. <pub-id pub-id-type="doi">10.1023/A:1022627411411</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Couprie</surname> <given-names>C.</given-names></name> <name><surname>Farabet</surname> <given-names>C.</given-names></name> <name><surname>Najman</surname> <given-names>L.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (<year>2013</year>). <source><italic>Indoor Semantic Segmentation Using Depth Information.</italic></source> Available online at: <ext-link ext-link-type="uri" xlink:href="https://hal.archives-ouvertes.fr/hal-00805105">https://hal.archives-ouvertes.fr/hal-00805105</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Da Costa</surname> <given-names>A. Z.</given-names></name> <name><surname>Figueroa</surname> <given-names>H. E. H.</given-names></name> <name><surname>Fracarolli</surname> <given-names>J. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Computer vision based detection of external defects on tomatoes using deep learning.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>190</volume> <fpage>131</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2019.12.003</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dashuta</surname> <given-names>A.</given-names></name> <name><surname>Klapp</surname> <given-names>I.</given-names></name></person-group> (<year>2018</year>). <source><italic>Melon Recognition in UAV Images to Estimate Yield of a Breeding Process.</italic></source> Available online at: <ext-link ext-link-type="uri" xlink:href="https://opg.optica.org/abstract.cfm?uri=EE-2018-ET4A.2">https://opg.optica.org/abstract.cfm?uri=EE-2018-ET4A.2</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Luna</surname> <given-names>R. G.</given-names></name> <name><surname>Dadios</surname> <given-names>E. P.</given-names></name> <name><surname>Bandala</surname> <given-names>A. A.</given-names></name> <name><surname>Vicerra</surname> <given-names>R. R. P.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Tomato fruit image dataset for deep transfer learning-based defect detection</article-title>,&#x201D; <source><italic>Proceedings of the 2019 IEEE International Conference on Cybernetics and Intelligent Systems (CIS) and IEEE Conference on Robotics, Automation and Mechatronics (RAM)</italic></source>, <publisher-loc>Bangkok</publisher-loc>, <fpage>356</fpage>&#x2013;<lpage>361</lpage>.</citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>H.</given-names></name> <name><surname>Zhu</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Recognition and counting of citrus flowers based on instance segmentation</article-title>. <source><italic>Nong Ye Gong Cheng Xue Bao</italic></source> <volume>36</volume>, <fpage>200</fpage>&#x2013;<lpage>207</lpage>. <pub-id pub-id-type="doi">10.11975/j.issn.1002-6819.2020.07.023</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dias</surname> <given-names>P. A.</given-names></name> <name><surname>Tabb</surname> <given-names>A.</given-names></name> <name><surname>Medeiros</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Apple flower detection using deep convolutional networks.</article-title> <source><italic>Comput. Ind.</italic></source> <volume>99</volume> <fpage>17</fpage>&#x2013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1016/j.compind.2018.03.010</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>W.</given-names></name> <name><surname>Roy</surname> <given-names>P.</given-names></name> <name><surname>Peng</surname> <given-names>C.</given-names></name> <name><surname>Isler</surname> <given-names>V.</given-names></name></person-group> (<year>2021</year>). <article-title>Ellipse R-CNN: learning to infer elliptical object from clustering and occlusion.</article-title> <source><italic>IEEE Trans. Image Process.</italic></source> <volume>30</volume> <fpage>2193</fpage>&#x2013;<lpage>2206</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2021.3050673</pub-id> <pub-id pub-id-type="pmid">33471755</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>Y.</given-names></name> <name><surname>Tao</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>W.</given-names></name> <name><surname>Ai</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning in aircraft design, dynamics, and control: review and prospects.</article-title> <source><italic>IEEE Trans. Aerospace Electron. Syst.</italic></source> <volume>57</volume> <fpage>2346</fpage>&#x2013;<lpage>2368</lpage>. <pub-id pub-id-type="doi">10.1109/TAES.2021.3056086</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Tian</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>On line detection of defective apples using computer vision system combined with deep learning methods.</article-title> <source><italic>J. Food Eng.</italic></source> <volume>286</volume>:<issue>110102</issue>. <pub-id pub-id-type="doi">10.1016/j.jfoodeng.2020.110102</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Farabet</surname> <given-names>C.</given-names></name> <name><surname>Couprie</surname> <given-names>C.</given-names></name> <name><surname>Najman</surname> <given-names>L.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (<year>2013</year>). <article-title>Learning hierarchical features for scene labeling.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>35</volume> <fpage>1915</fpage>&#x2013;<lpage>1929</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2012.231</pub-id> <pub-id pub-id-type="pmid">23787344</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Farjon</surname> <given-names>G.</given-names></name> <name><surname>Krikeb</surname> <given-names>O.</given-names></name> <name><surname>Hillel</surname> <given-names>A. B.</given-names></name> <name><surname>Alchanatis</surname> <given-names>V.</given-names></name></person-group> (<year>2020</year>). <article-title>Detection and counting of flowers on apple trees for better chemical thinning decisions.</article-title> <source><italic>Precis. Agric.</italic></source> <volume>21</volume> <fpage>503</fpage>&#x2013;<lpage>521</lpage>. <pub-id pub-id-type="doi">10.1007/s11119-019-09679-1</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fawcett</surname> <given-names>T.</given-names></name></person-group> (<year>2006</year>). <article-title>An introduction to ROC analysis</article-title>. <source><italic>Pattern Recognit. Lett</italic></source>. <volume>27</volume> <fpage>861</fpage>&#x2013;<lpage>874</lpage>.</citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Feng</surname> <given-names>Y.</given-names></name> <name><surname>Majeed</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Kiwifruit detection in field images using faster R-CNN with ZFNet.</article-title> <source><italic>IFAC Papersonline</italic></source> <volume>51</volume> <fpage>45</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1016/j.ifacol.2018.08.059</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Feng</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Gao</surname> <given-names>F.</given-names></name> <name><surname>Majeed</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Fast and accurate detection of kiwifruit in orchard using improved YOLOv3-tiny model.</article-title> <source><italic>Precis. Agric.</italic></source> <volume>22</volume> <fpage>754</fpage>&#x2013;<lpage>776</lpage>. <pub-id pub-id-type="doi">10.1007/s11119-020-09754-y</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Gao</surname> <given-names>F.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name></person-group> (<year>2020a</year>). <article-title>Application of consumer RGB-D cameras for fruit detection and localization in field: a critical review.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>177</volume>:<issue>105687</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105687</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Majeed</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name></person-group> (<year>2020b</year>). <article-title>Faster R&#x2013;CNN&#x2013;based apple detection in dense-foliage fruiting-wall trees using RGB and depth features for robotic harvesting.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>197</volume> <fpage>245</fpage>&#x2013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2020.07.007</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fukushima</surname> <given-names>K.</given-names></name></person-group> (<year>1980</year>). <article-title>Neocognitron: a self-organizing neural network model for a mechanism of pattern recognition unaffected by shift in position.</article-title> <source><italic>Biol. Cybern.</italic></source> <volume>36</volume> <fpage>193</fpage>&#x2013;<lpage>202</lpage>. <pub-id pub-id-type="doi">10.1007/BF00344251</pub-id> <pub-id pub-id-type="pmid">7370364</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gai</surname> <given-names>R.</given-names></name> <name><surname>Chen</surname> <given-names>N.</given-names></name> <name><surname>Yuan</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>A detection algorithm for cherry fruits based on the improved YOLO-v4 model.</article-title> <source><italic>Neural Comput. Appl.</italic></source> <pub-id pub-id-type="doi">10.1007/s00521-021-06029-z</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ganesh</surname> <given-names>P.</given-names></name> <name><surname>Volle</surname> <given-names>K.</given-names></name> <name><surname>Burks</surname> <given-names>T. F.</given-names></name> <name><surname>Mehta</surname> <given-names>S. S.</given-names></name></person-group> (<year>2019</year>). <article-title>Deep orange: Mask R-CNN based orange detection and segmentation.</article-title> <source><italic>IFAC Papersonline</italic></source> <volume>52</volume> <fpage>70</fpage>&#x2013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1016/j.ifacol.2019.12.499</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>F.</given-names></name> <name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Majeed</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Multi-class fruit-on-plant detection for apple in SNAP system using faster R-CNN.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>176</volume>:<issue>105634</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105634</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ge</surname> <given-names>Y.</given-names></name> <name><surname>Xiong</surname> <given-names>Y.</given-names></name> <name><surname>Tenorio</surname> <given-names>G. L.</given-names></name> <name><surname>From</surname> <given-names>P. J.</given-names></name></person-group> (<year>2019</year>). <article-title>Fruit localization and environment perception for strawberry harvesting robots.</article-title> <source><italic>IEEE Access</italic></source> <volume>7</volume> <fpage>147642</fpage>&#x2013;<lpage>147652</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2946369</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gen&#x00E9;-Mola</surname> <given-names>J.</given-names></name> <name><surname>Gregorio</surname> <given-names>E.</given-names></name> <name><surname>Guevara</surname> <given-names>J.</given-names></name> <name><surname>Auat</surname> <given-names>F.</given-names></name> <name><surname>Sanz-Cortiella</surname> <given-names>R.</given-names></name> <name><surname>Escol&#x00E0;</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2019a</year>). <article-title>Fruit detection in an apple orchard using a mobile terrestrial laser scanner.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>187</volume> <fpage>171</fpage>&#x2013;<lpage>184</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2019.08.017</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gen&#x00E9;-Mola</surname> <given-names>J.</given-names></name> <name><surname>Vilaplana</surname> <given-names>V.</given-names></name> <name><surname>Rosell-Polo</surname> <given-names>J. R.</given-names></name> <name><surname>Morros</surname> <given-names>J.</given-names></name> <name><surname>Ruiz-Hidalgo</surname> <given-names>J.</given-names></name> <name><surname>Gregorio</surname> <given-names>E.</given-names></name></person-group> (<year>2019b</year>). <article-title>Multi-modal deep learning for fuji apple detection using RGB-D cameras and their radiometric capabilities.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>162</volume> <fpage>689</fpage>&#x2013;<lpage>698</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.05.016</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gen&#x00E9;-Mola</surname> <given-names>J.</given-names></name> <name><surname>Sanz-Cortiella</surname> <given-names>R.</given-names></name> <name><surname>Rosell-Polo</surname> <given-names>J. R.</given-names></name> <name><surname>Morros</surname> <given-names>J.</given-names></name> <name><surname>Ruiz-Hidalgo</surname> <given-names>J.</given-names></name> <name><surname>Vilaplana</surname> <given-names>V.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Fruit detection and 3D location using instance segmentation neural networks and structure-from-motion photogrammetry.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>169</volume>:<issue>105165</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.105165</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). <source><italic>fast r-cnn.</italic></source> Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cv-foundation.org/openaccess/content_iccv_2015/papers/Girshick_Fast_R-CNN_ICCV_2015_paper.pdf">https://www.cv-foundation.org/openaccess/content_iccv_2015/papers/Girshick_Fast_R-CNN_ICCV_2015_paper.pdf</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Donahue</surname> <given-names>J.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name> <name><surname>Malik</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Rich feature hierarchies for accurate object detection and semantic segmentation</article-title>,&#x201D; in <source><italic>Proceedings of the 2014 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Columbus, OH</publisher-loc>, <fpage>580</fpage>&#x2013;<lpage>587</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2014.81</pub-id></citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Glorot</surname> <given-names>X.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2010</year>). &#x201C;<article-title>Understanding the difficulty of training deep feed forward neural networks</article-title>,&#x201D; in <source><italic>Proceedings of the 13th 2010 International Conference on Artificial Intelligence and Statistics</italic></source>, <volume>Vol. 9</volume>, <publisher-loc>Sardinia</publisher-loc>, <fpage>249</fpage>&#x2013;<lpage>256</lpage>.</citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gongal</surname> <given-names>A.</given-names></name> <name><surname>Silwal</surname> <given-names>A.</given-names></name> <name><surname>Amatya</surname> <given-names>S.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <name><surname>Lewis</surname> <given-names>K.</given-names></name></person-group> (<year>2016</year>). <article-title>Apple crop-load estimation with over-the-row machine vision system.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>120</volume> <fpage>26</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2015.10.022</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gonzalez</surname> <given-names>S.</given-names></name> <name><surname>Arellano</surname> <given-names>C.</given-names></name> <name><surname>Tapia</surname> <given-names>J. E.</given-names></name></person-group> (<year>2019</year>). <article-title>Deepblueberry: quantification of blueberries in the wild using instance segmentation.</article-title> <source><italic>IEEE Access</italic></source> <volume>7</volume> <fpage>105776</fpage>&#x2013;<lpage>105788</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2933062</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname> <given-names>A.</given-names></name> <name><surname>Harrison</surname> <given-names>P. J.</given-names></name> <name><surname>Wieslander</surname> <given-names>H.</given-names></name> <name><surname>Pielawski</surname> <given-names>N.</given-names></name> <name><surname>Kartasalo</surname> <given-names>K.</given-names></name> <name><surname>Partel</surname> <given-names>G.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Deep learning in image cytometry: a review.</article-title> <source><italic>Cytometry A</italic></source> <volume>95</volume> <fpage>366</fpage>&#x2013;<lpage>380</lpage>. <pub-id pub-id-type="doi">10.1002/cyto.a.23701</pub-id> <pub-id pub-id-type="pmid">30565841</pub-id></citation></ref>
<ref id="B54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hariharan</surname> <given-names>B.</given-names></name> <name><surname>Arbel&#x00E1;ez</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Malik</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Simultaneous detection and segmentation</article-title>,&#x201D; in <source><italic>Computer Vision &#x2013; ECCV 2014. ECCV 2014. Lecture Notes in Computer Science</italic></source>, <volume>Vol. 8695</volume> <role>eds</role> <person-group person-group-type="editor"><name><surname>Fleet</surname> <given-names>D.</given-names></name> <name><surname>Pajdla</surname> <given-names>T.</given-names></name> <name><surname>Schiele</surname> <given-names>B.</given-names></name> <name><surname>Tuytelaars</surname> <given-names>T.</given-names></name></person-group> <publisher-loc>(Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>297</fpage>&#x2013;<lpage>312</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-10584-0_20</pub-id></citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Gkioxari</surname> <given-names>G.</given-names></name> <name><surname>Dollar</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <source><italic>mask r-cnn.</italic></source> Available online at: <ext-link ext-link-type="uri" xlink:href="https://openaccess.thecvf.com/content_ICCV_2017/papers/He_Mask_R-CNN_ICCV_2017_paper.pdf">https://openaccess.thecvf.com/content_ICCV_2017/papers/He_Mask_R-CNN_ICCV_2017_paper.pdf</ext-link></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). &#x201C;<article-title>Delving deep into rectifiers: surpassing human-level performance on ImageNet classification</article-title>,&#x201D; in <source><italic>Proceedings of the International Conference on Computer Vision</italic></source>, <publisher-loc>Santiago</publisher-loc>, <fpage>1026</fpage>&#x2013;<lpage>1034</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2015.123</pub-id></citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K. M.</given-names></name> <name><surname>Zhang</surname> <given-names>X. Y.</given-names></name> <name><surname>Ren</surname> <given-names>S. Q.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Deep residual learning for image recognition</article-title>,&#x201D; in <source><italic>Proceedings of the 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Las Vegas, NV</publisher-loc>, <fpage>770</fpage>&#x2013;<lpage>778</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>L.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name></person-group> (<year>2017</year>). <article-title>Effect of fruit location on apple detachment with mechanical shaking.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>157</volume> <fpage>63</fpage>&#x2013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2017.02.009</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Bag of tricks for image classification with convolutional neural networks</article-title>,&#x201D; in <source><italic>Proceedings of the 2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Long Beach, CA</publisher-loc>.</citation></ref>
<ref id="B60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G. E.</given-names></name> <name><surname>Osindero</surname> <given-names>S.</given-names></name> <name><surname>Teh</surname> <given-names>Y.</given-names></name></person-group> (<year>2006</year>). <article-title>A fast learning algorithm for deep belief nets.</article-title> <source><italic>Neural Comput.</italic></source> <volume>18</volume> <fpage>1527</fpage>&#x2013;<lpage>1554</lpage>. <pub-id pub-id-type="doi">10.1162/neco.2006.18.7.1527</pub-id> <pub-id pub-id-type="pmid">16764513</pub-id></citation></ref>
<ref id="B61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Howard</surname> <given-names>A. G.</given-names></name> <name><surname>Zhu</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>B.</given-names></name> <name><surname>Kalenichenko</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Weyand</surname> <given-names>T.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>MobileNets: efficient convolutional neural networks for mobile vision applications.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1704.04861">http://arxiv.org/abs/1704.04861</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>G.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Van Der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Densely connected convolutional networks</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 30th IEEE conference on computer vision and pattern recognition (CVPR)</italic></source>, <publisher-loc>Honolulu, HI</publisher-loc>, <fpage>2261</fpage>&#x2013;<lpage>2269</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.243</pub-id></citation></ref>
<ref id="B63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Huang</surname> <given-names>T.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Lyu</surname> <given-names>S.</given-names></name> <name><surname>Hong</surname> <given-names>T.</given-names></name></person-group> (<year>2022</year>). <article-title>Design of citrus fruit detection system based on mobile platform and edge computer device.</article-title> <source><italic>Sensors</italic></source> <volume>22</volume>:<issue>59</issue>. <pub-id pub-id-type="doi">10.3390/s22010059</pub-id> <pub-id pub-id-type="pmid">35009602</pub-id></citation></ref>
<ref id="B64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Basanta</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Using fuzzy mask R-CNN model to automatically identify tomato ripeness.</article-title> <source><italic>IEEE Access</italic></source> <volume>8</volume> <fpage>207672</fpage>&#x2013;<lpage>207682</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.3038184</pub-id></citation></ref>
<ref id="B65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Z.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Gong</surname> <given-names>Y.</given-names></name> <name><surname>Huang</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Mask Scoring R-CNN</article-title>,&#x201D; in <source><italic>Proceedings of the 2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Long Beach, CA</publisher-loc>.</citation></ref>
<ref id="B66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jahanbakhshi</surname> <given-names>A.</given-names></name> <name><surname>Momeny</surname> <given-names>M.</given-names></name> <name><surname>Mahmoudi</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Classification of sour lemons based on apparent defects using stochastic pooling mechanism in deep convolutional neural networks.</article-title> <source><italic>Sci. Hortic.</italic></source> <volume>263</volume>:<issue>109133</issue>. <pub-id pub-id-type="doi">10.1016/j.scienta.2019.109133</pub-id></citation></ref>
<ref id="B67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Janowski</surname> <given-names>A.</given-names></name> <name><surname>Ka&#x017A;mierczak</surname> <given-names>R.</given-names></name> <name><surname>Kowalczyk</surname> <given-names>C.</given-names></name> <name><surname>Szulwic</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Detecting apples in the wild: potential for harvest quantity estimation.</article-title> <source><italic>Sustainability</italic></source> <volume>13</volume>:<issue>8054</issue>. <pub-id pub-id-type="doi">10.3390/su13148054</pub-id></citation></ref>
<ref id="B68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>W.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name> <name><surname>Luo</surname> <given-names>R.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Lian</surname> <given-names>J.</given-names></name> <name><surname>Zheng</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Detection and segmentation of overlapped fruits based on optimized mask R-CNN application in apple harvesting robot.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>172</volume>:<issue>105380</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105380</pub-id></citation></ref>
<ref id="B69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Shelhamer</surname> <given-names>E.</given-names></name> <name><surname>Donahue</surname> <given-names>J.</given-names></name> <name><surname>Karayev</surname> <given-names>S.</given-names></name> <name><surname>Long</surname> <given-names>J.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2014</year>). &#x201C;<article-title>Caffe: convolutional architecture for fast feature embedding</article-title>,&#x201D; in <source><italic>Proceedings of the 2014 22nd ACM international conference on Multimedia</italic></source>, <publisher-loc>New York; NY</publisher-loc>.</citation></ref>
<ref id="B70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>B.</given-names></name> <name><surname>He</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>T.</given-names></name> <name><surname>Song</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Fusion of machine vision technology and AlexNet-CNNs deep learning network for the detection of postharvest apple pesticide residues.</article-title> <source><italic>Artif. Intell. Agric.</italic></source> <volume>1</volume> <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/j.aiia.2019.02.001</pub-id></citation></ref>
<ref id="B71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jie</surname> <given-names>D.</given-names></name> <name><surname>Wu</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Ye</surname> <given-names>D.</given-names></name> <name><surname>Wei</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>Research on citrus grandis granulation determination based on hyperspectral imaging through deep learning.</article-title> <source><italic>Food Anal. Methods</italic></source> <volume>14</volume> <fpage>280</fpage>&#x2013;<lpage>289</lpage>. <pub-id pub-id-type="doi">10.1007/s12161-020-01873-6</pub-id></citation></ref>
<ref id="B72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Joe</surname> <given-names>G. G.</given-names></name> <name><surname>Shaun</surname> <given-names>M. K.</given-names></name> <name><surname>Lewis</surname> <given-names>M.</given-names></name> <name><surname>David</surname> <given-names>T. J.</given-names></name></person-group> (<year>2022</year>). <article-title>A guide to machine learning for biologists.</article-title> <source><italic>Nat. Rev. Mol. Cell Biol.</italic></source> <volume>23</volume> <fpage>40</fpage>&#x2013;<lpage>55</lpage>.</citation></ref>
<ref id="B73"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kalantar</surname> <given-names>A.</given-names></name> <name><surname>Dashuta</surname> <given-names>A.</given-names></name> <name><surname>Edan</surname> <given-names>Y.</given-names></name> <name><surname>Dafna</surname> <given-names>A.</given-names></name> <name><surname>Gur</surname> <given-names>A.</given-names></name> <name><surname>Klapp</surname> <given-names>I.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Estimating melon yield for breeding processes by machine-vision processing of UAV images</article-title>,&#x201D; in <source><italic>Proceedings of the 12th European Conference on Precision Agriculture, ECPA 2019)</italic></source>, <role>ed.</role> <person-group person-group-type="editor"><name><surname>Stafford</surname> <given-names>J. V.</given-names></name></person-group> (<publisher-loc>Wageningen</publisher-loc>: <publisher-name>Wageningen Academic Publishers</publisher-name>), <fpage>381</fpage>&#x2013;<lpage>387</lpage>. <pub-id pub-id-type="doi">10.3920/978-90-8686-888-9_47</pub-id></citation></ref>
<ref id="B74"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kalantar</surname> <given-names>A.</given-names></name> <name><surname>Edan</surname> <given-names>Y.</given-names></name> <name><surname>Gur</surname> <given-names>A.</given-names></name> <name><surname>Klapp</surname> <given-names>I.</given-names></name></person-group> (<year>2020</year>). <article-title>A deep learning system for single and overall weight estimation of melons using unmanned aerial vehicle images.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>178</volume>:<issue>105748</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105748</pub-id></citation></ref>
<ref id="B75"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kalogerakis</surname> <given-names>E.</given-names></name> <name><surname>Averkiou</surname> <given-names>M.</given-names></name> <name><surname>Maji</surname> <given-names>S.</given-names></name> <name><surname>Chaudhuri</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>3D shape segmentation with projective convolutional networks</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Honolulu, HI</publisher-loc>.</citation></ref>
<ref id="B76"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kamilaris</surname> <given-names>A.</given-names></name> <name><surname>Prenafeta-Bold&#x00FA;</surname> <given-names>F. X.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep learning in agriculture: a survey.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>147</volume> <fpage>70</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2018.02.016</pub-id></citation></ref>
<ref id="B77"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>C.</given-names></name></person-group> (<year>2020</year>). <article-title>Fast implementation of real-time fruit detection in apple orchards using deep learning.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>168</volume>:<issue>105108</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.105108</pub-id></citation></ref>
<ref id="B78"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khosravi</surname> <given-names>H.</given-names></name> <name><surname>Saedi</surname> <given-names>S. I.</given-names></name> <name><surname>Rezaei</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Real-time recognition of on-branch olive ripening stages by a deep convolutional neural network.</article-title> <source><italic>Sci. Hortic.</italic></source> <volume>287</volume>:<issue>110252</issue>. <pub-id pub-id-type="doi">10.1016/j.scienta.2021.110252</pub-id></citation></ref>
<ref id="B79"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koirala</surname> <given-names>A.</given-names></name> <name><surname>Walsh</surname> <given-names>K. B.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>McCarthy</surname> <given-names>C.</given-names></name></person-group> (<year>2019b</year>). <article-title>Deep learning &#x2013; method overview and review of use for fruit detection and yield estimation.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>162</volume> <fpage>219</fpage>&#x2013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.04.017</pub-id></citation></ref>
<ref id="B80"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koirala</surname> <given-names>A.</given-names></name> <name><surname>Walsh</surname> <given-names>K. B.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>McCarthy</surname> <given-names>C.</given-names></name></person-group> (<year>2019a</year>). <article-title>Deep learning for real-time fruit detection and orchard fruit load estimation: benchmarking of &#x2018;MangoYOLO&#x2019;.</article-title> <source><italic>Precis. Agric.</italic></source> <volume>20</volume> <fpage>1107</fpage>&#x2013;<lpage>1135</lpage>. <pub-id pub-id-type="doi">10.1007/s11119-019-09642-0</pub-id></citation></ref>
<ref id="B81"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2017</year>). <article-title>ImageNet classification with deep convolutional neural networks.</article-title> <source><italic>Commun. ACM</italic></source> <volume>60</volume> <fpage>84</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1145/3065386</pub-id></citation></ref>
<ref id="B82"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lamb</surname> <given-names>N.</given-names></name> <name><surname>Chuah</surname> <given-names>M. C.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>A strawberry detection system using convolutional neural networks</article-title>,&#x201D; in <source><italic>Proceedings of the 2018 IEEE International Conference on Big Data(Big Data)</italic></source>, <publisher-loc>Seattle, WA</publisher-loc>, <fpage>2515</fpage>&#x2013;<lpage>2520</lpage>. <pub-id pub-id-type="doi">10.1109/BigData.2018.8622466</pub-id></citation></ref>
<ref id="B83"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>1995</year>). &#x201C;<article-title>Convolutional networks for images</article-title>,&#x201D; in <source><italic>Speech, and Time-Series. Handbook of Brain Theory &#x0026; Neural Networks</italic></source>, <role>ed.</role> <person-group person-group-type="editor"><name><surname>Arbib</surname> <given-names>M. A.</given-names></name></person-group> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>.</citation></ref>
<ref id="B84"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Boser</surname> <given-names>B.</given-names></name> <name><surname>Denker</surname> <given-names>J. S.</given-names></name> <name><surname>Henderson</surname> <given-names>D.</given-names></name> <name><surname>Howard</surname> <given-names>R. E.</given-names></name> <name><surname>Hubbard</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>1989</year>). <article-title>Backpropagation applied to handwritten zip code recognition.</article-title> <source><italic>Neural Comput.</italic></source> <volume>1</volume> <fpage>541</fpage>&#x2013;<lpage>551</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1989.1.4.541</pub-id></citation></ref>
<ref id="B85"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Bottou</surname> <given-names>L.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Haffner</surname> <given-names>P.</given-names></name></person-group> (<year>1998</year>). <article-title>Gradient-based learning applied to document recognition.</article-title> <source><italic>Proc. IEEE</italic></source> <volume>86</volume> <fpage>2278</fpage>&#x2013;<lpage>2324</lpage>. <pub-id pub-id-type="doi">10.1109/9780470544976.ch9</pub-id></citation></ref>
<ref id="B86"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Chesser</surname> <given-names>J.</given-names></name> <name><surname>Gary</surname> <given-names>D.</given-names></name> <name><surname>Purswell</surname> <given-names>J. L.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Practices and applications of convolutional neural network-based computer vision systems in animal farming: a review.</article-title> <source><italic>Sensors</italic></source> <volume>21</volume>:<issue>1492</issue>. <pub-id pub-id-type="doi">10.3390/s21041492</pub-id> <pub-id pub-id-type="pmid">33670030</pub-id></citation></ref>
<ref id="B87"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>G.</given-names></name> <name><surname>Tang</surname> <given-names>Y.</given-names></name> <name><surname>Zou</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Xiong</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>In-field citrus detection and localisation based on RGB-D image analysis.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>186</volume> <fpage>34</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2019.06.019</pub-id></citation></ref>
<ref id="B88"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>G.</given-names></name> <name><surname>Tang</surname> <given-names>Y.</given-names></name> <name><surname>Zou</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Three-dimensional reconstruction of guava fruits and branches using instance segmentation and geometry analysis.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>184</volume>:<issue>106107</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2021.106107</pub-id></citation></ref>
<ref id="B89"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>P.</given-names></name> <name><surname>Lee</surname> <given-names>W. S.</given-names></name> <name><surname>Chen</surname> <given-names>Y. M.</given-names></name> <name><surname>Peres</surname> <given-names>N.</given-names></name> <name><surname>Fraisse</surname> <given-names>C.</given-names></name></person-group> (<year>2020</year>). <article-title>A deep-level region-based visual representation architecture for detecting strawberry flowers in an outdoor field.</article-title> <source><italic>Precis. Agric.</italic></source> <volume>21</volume> <fpage>387</fpage>&#x2013;<lpage>402</lpage>. <pub-id pub-id-type="doi">10.1007/s11119-019-09673-7</pub-id></citation></ref>
<ref id="B90"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Jia</surname> <given-names>J.</given-names></name> <name><surname>Fidler</surname> <given-names>S.</given-names></name> <name><surname>Urtasun</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>SGN: sequential grouping networks for instance segmentation</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 IEEE International Conference on Computer Vision (ICCV)</italic></source>, <publisher-loc>Venice</publisher-loc>, <fpage>3516</fpage>&#x2013;<lpage>3524</lpage>.</citation></ref>
<ref id="B91"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Qi</surname> <given-names>L.</given-names></name> <name><surname>Qin</surname> <given-names>H.</given-names></name> <name><surname>Shi</surname> <given-names>J.</given-names></name> <name><surname>Jia</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Path aggregation network for instance segmentation</article-title>,&#x201D; in <source><italic>Proceedings of the 2018 IEEE Conference on Computer Vision and Pattern</italic></source>, <publisher-loc>Salt Lake City, UT</publisher-loc>.</citation></ref>
<ref id="B92"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Qi</surname> <given-names>X.</given-names></name> <name><surname>Shi</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Jia</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Mufti-scale Patch aggregation(MPA)for Simultaneous Detection and Segmentation}G]</article-title>,&#x201D; in <source><italic>Proceedings of the 2016 IEEE Conference on Computer Vision and Pattern Recognition</italic></source>, <publisher-loc>Las Vegas, NV</publisher-loc>, <fpage>3141</fpage>&#x2013;<lpage>3149</lpage>.</citation></ref>
<ref id="B93"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Anguelov</surname> <given-names>D.</given-names></name> <name><surname>Erhan</surname> <given-names>D.</given-names></name> <name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Reed</surname> <given-names>S.</given-names></name> <name><surname>Fu</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2016</year>). &#x201C;<article-title>SSD: single shot MultiBox detector</article-title>,&#x201D; in <source><italic>Computer Vision &#x2013; ECCV 2016. ECCV 2016. Lecture Notes in Computer Science</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Leibe</surname> <given-names>B.</given-names></name> <name><surname>Matas</surname> <given-names>J.</given-names></name> <name><surname>Sebe</surname> <given-names>N.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer)</publisher-name>, <fpage>21</fpage>&#x2013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-46448-0_2</pub-id></citation></ref>
<ref id="B94"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y. P.</given-names></name> <name><surname>Yang</surname> <given-names>C.</given-names></name> <name><surname>Ling</surname> <given-names>H.</given-names></name> <name><surname>Mabu</surname> <given-names>S.</given-names></name> <name><surname>Kuremoto</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>A visual system of citrus picking robot using convolutional neural networks</article-title>,&#x201D; in <source><italic>Proceedings of the 2018 5th International Conference on Systems and Informatics (ICSAI)</italic></source>, <publisher-loc>Nanjing</publisher-loc>, <fpage>344</fpage>&#x2013;<lpage>349</lpage>. <pub-id pub-id-type="doi">10.1109/ICSAI.2018.8599325</pub-id></citation></ref>
<ref id="B95"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Majeed</surname> <given-names>Y.</given-names></name> <name><surname>Feng</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Improved kiwifruit detection using pre-trained VGG16 with RGB and NIR information fusion.</article-title> <source><italic>IEEE Access</italic></source> <volume>8</volume> <fpage>2327</fpage>&#x2013;<lpage>2336</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2962513</pub-id></citation></ref>
<ref id="B96"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Longye</surname> <given-names>X.</given-names></name> <name><surname>Zhuo</surname> <given-names>W.</given-names></name> <name><surname>Haishen</surname> <given-names>L.</given-names></name> <name><surname>Xilong</surname> <given-names>K.</given-names></name> <name><surname>Changhui</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Overlapping citrus segmentation and reconstruction based on mask R-CNN model and concave region simplification and distance analysis.</article-title> <source><italic>J. Phys. Conf. Ser.</italic></source> <volume>1345</volume>:<issue>32064</issue>. <pub-id pub-id-type="doi">10.1088/1742-6596/1345/3/032064</pub-id></citation></ref>
<ref id="B97"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Majeed</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Deep learning based segmentation for automated training of apple trees on trellis wires.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>170</volume>:<issue>105277</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105277</pub-id></citation></ref>
<ref id="B98"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mohsen</surname> <given-names>Y. N.</given-names></name> <name><surname>Dan</surname> <given-names>T.</given-names></name> <name><surname>Milad</surname> <given-names>E.</given-names></name></person-group> (<year>2021</year>). <article-title>Using hybrid artificial intelligence and evolutionary optimization algorithms for estimating soybean yield and fresh biomass using hyperspectral vegetation indices.</article-title> <source><italic>Remote Sens.</italic></source> <volume>13</volume>:<issue>2555</issue>. <pub-id pub-id-type="doi">10.3390/rs13132555</pub-id></citation></ref>
<ref id="B99"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Momeny</surname> <given-names>M.</given-names></name> <name><surname>Jahanbakhshi</surname> <given-names>A.</given-names></name> <name><surname>Jafarnezhad</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Accurate classification of cherry fruit using deep CNN based on hybrid pooling approach.</article-title> <source><italic>Postharvest Biol. Technol.</italic></source> <volume>166</volume>:<issue>111204</issue>. <pub-id pub-id-type="doi">10.1016/j.postharvbio.2020.111204</pub-id></citation></ref>
<ref id="B100"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mu</surname> <given-names>L.</given-names></name> <name><surname>Gao</surname> <given-names>Z.</given-names></name> <name><surname>Cui</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Fu</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>Kiwifruit detection of far-view and occluded fruit based on improved AlexNet.</article-title> <source><italic>Trans. Chin. Soc. Agric. Mach</italic>.</source> <volume>50</volume> <fpage>24</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2019.10.003</pub-id></citation></ref>
<ref id="B101"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mu</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Ninomiya</surname> <given-names>S.</given-names></name> <name><surname>Guo</surname> <given-names>W.</given-names></name></person-group> (<year>2020</year>). <article-title>Intact detection of highly occluded immature tomatoes on plants using deep learning techniques.</article-title> <source><italic>Sensors</italic></source> <volume>20</volume>:<issue>2984</issue>. <pub-id pub-id-type="doi">10.3390/s20102984</pub-id> <pub-id pub-id-type="pmid">32466108</pub-id></citation></ref>
<ref id="B102"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Naranjo-Torres</surname> <given-names>J.</given-names></name> <name><surname>Mora</surname> <given-names>M.</given-names></name> <name><surname>Hern&#x00E1;ndez-Garc&#x00ED;a</surname> <given-names>R.</given-names></name> <name><surname>Barrientos</surname> <given-names>R. J.</given-names></name> <name><surname>Fredes</surname> <given-names>C.</given-names></name> <name><surname>Valenzuela</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>A review of convolutional neural network applied to fruit image processing.</article-title> <source><italic>Appl. Sci.</italic></source> <volume>10</volume>:<issue>3443</issue>. <pub-id pub-id-type="doi">10.3390/app10103443</pub-id></citation></ref>
<ref id="B103"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>H.</given-names></name> <name><surname>Kieu</surname> <given-names>L.</given-names></name> <name><surname>Wen</surname> <given-names>T.</given-names></name> <name><surname>Cai</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep learning methods in transportation domain: a review.</article-title> <source><italic>IET Intell. Transp. Syst.</italic></source> <volume>12</volume> <fpage>998</fpage>&#x2013;<lpage>1004</lpage>. <pub-id pub-id-type="doi">10.1049/iet-its.2018.0064</pub-id></citation></ref>
<ref id="B104"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>T. T.</given-names></name> <name><surname>Vandevoorde</surname> <given-names>K.</given-names></name> <name><surname>Wouters</surname> <given-names>N.</given-names></name> <name><surname>Kayacan</surname> <given-names>E.</given-names></name> <name><surname>De Baerdemaeker</surname> <given-names>J. G.</given-names></name> <name><surname>Saeys</surname> <given-names>W.</given-names></name></person-group> (<year>2016</year>). <article-title>Detection of red and bicoloured apples on tree with an RGB-D camera.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>146</volume> <fpage>33</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2016.01.007</pub-id></citation></ref>
<ref id="B105"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Takeda</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep learning image segmentation and extraction of blueberry fruit traits associated with harvestability and yield.</article-title> <source><italic>Hortic. Res.</italic></source> <volume>7</volume>:<issue>110</issue>. <pub-id pub-id-type="doi">10.1038/s41438-020-0323-3</pub-id> <pub-id pub-id-type="pmid">32637138</pub-id></citation></ref>
<ref id="B106"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Takeda</surname> <given-names>F.</given-names></name></person-group> (<year>2021</year>). <article-title>Three-dimensional photogrammetry with deep learning instance segmentation to extract berry fruit harvestability traits.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>171</volume> <fpage>297</fpage>&#x2013;<lpage>309</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2020.11.010</pub-id></citation></ref>
<ref id="B107"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nur Alam</surname> <given-names>M.</given-names></name> <name><surname>Saugat</surname> <given-names>S.</given-names></name> <name><surname>Santosh</surname> <given-names>D.</given-names></name> <name><surname>Sarkar</surname> <given-names>M. I.</given-names></name> <name><surname>Al-Absi</surname> <given-names>A. A.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>Apple defect detection Q64 based on deep convolutional neural network</article-title>,&#x201D; in <source><italic>Proceedings of the 2020 International Conference on Smart Computing and Cyber Security: Strategic Foresight, Security Challenges and Innovation</italic></source>, <role>ed.</role> <person-group person-group-type="editor"><name><surname>Pattnaik</surname> <given-names>P. K.</given-names></name></person-group> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Verlag</publisher-name>), <fpage>215</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1007/978-981-15-7990-5_21</pub-id></citation></ref>
<ref id="B108"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nyarko</surname> <given-names>E. K.</given-names></name> <name><surname>Vidovi&#x0107;</surname> <given-names>I.</given-names></name> <name><surname>Rado&#x010D;aj</surname> <given-names>K.</given-names></name> <name><surname>Cupec</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>A nearest neighbor approach for fruit recognition in RGB-D images based on detection of convex surfaces.</article-title> <source><italic>Expert Syst. Appl.</italic></source> <volume>114</volume> <fpage>454</fpage>&#x2013;<lpage>466</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2018.07.048</pub-id></citation></ref>
<ref id="B109"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Okamoto</surname> <given-names>H.</given-names></name> <name><surname>Lee</surname> <given-names>W. S.</given-names></name></person-group> (<year>2009</year>). <article-title>Green citrus detection using hyperspectral imaging</article-title>. <source><italic>Comput. Electron. Agric.</italic></source> <volume>66</volume>, <fpage>201</fpage>&#x2013;<lpage>208</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2009.02.004</pub-id></citation></ref>
<ref id="B110"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Onishi</surname> <given-names>Y.</given-names></name> <name><surname>Yoshida</surname> <given-names>T.</given-names></name> <name><surname>Kurita</surname> <given-names>H.</given-names></name> <name><surname>Fukao</surname> <given-names>T.</given-names></name> <name><surname>Arihara</surname> <given-names>H.</given-names></name> <name><surname>Iwai</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>An automated fruit harvesting robot by using deep learning.</article-title> <source><italic>ROBOMECH J.</italic></source> <volume>6</volume> <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1186/s40648-019-0141-2</pub-id></citation></ref>
<ref id="B111"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oquab</surname> <given-names>M.</given-names></name> <name><surname>Bottou</surname> <given-names>L.</given-names></name> <name><surname>Laptev</surname> <given-names>I.</given-names></name> <name><surname>Sivic</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Learning and transferring mid-level image representations using convolutional neural networks</article-title>,&#x201D; in <source><italic>Proceedings of the 2014 Computer Vision &#x0026; Pattern Recognition</italic></source>, <publisher-loc>Columbus, OH</publisher-loc>.</citation></ref>
<ref id="B112"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pal</surname> <given-names>N. R.</given-names></name> <name><surname>Pal</surname> <given-names>S. K.</given-names></name></person-group> (<year>1993</year>). <article-title>A review on image segmentation techniques.</article-title> <source><italic>Pattern Recogn.</italic></source> <volume>26</volume> <fpage>1277</fpage>&#x2013;<lpage>1294</lpage>. <pub-id pub-id-type="doi">10.1016/0031-3203(93)90135-J</pub-id></citation></ref>
<ref id="B113"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Palacios</surname> <given-names>F.</given-names></name> <name><surname>Bueno</surname> <given-names>G.</given-names></name> <name><surname>Salido</surname> <given-names>J.</given-names></name> <name><surname>Diago</surname> <given-names>M. P.</given-names></name> <name><surname>Hern&#x00E1;ndez</surname> <given-names>I.</given-names></name> <name><surname>Tardaguila</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Automated grapevine flower detection and quantification method based on computer vision and deep learning from on-the-go imaging using a mobile sensing platform under field conditions.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>178</volume>:<issue>105796</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105796</pub-id></citation></ref>
<ref id="B114"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parvathi</surname> <given-names>S.</given-names></name> <name><surname>Tamil Selvi</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Detection of maturity stages of coconuts in complex background using faster R-CNN model.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>202</volume> <fpage>119</fpage>&#x2013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2020.12.002</pub-id></citation></ref>
<ref id="B115"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paszke</surname> <given-names>A.</given-names></name> <name><surname>Gross</surname> <given-names>S.</given-names></name> <name><surname>Massa</surname> <given-names>F.</given-names></name> <name><surname>Lerer</surname> <given-names>A.</given-names></name> <name><surname>Bradbury</surname> <given-names>J.</given-names></name> <name><surname>Chanan</surname> <given-names>G.</given-names></name><etal/></person-group> (<year>2019</year>). &#x201C;<article-title>PyTorch: an imperative style, high-performance deep learning library</article-title>,&#x201D; in <source><italic>Advances in Neural Information Processing Systems 32</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Wallach</surname> <given-names>H.</given-names></name> <name><surname>Larochelle</surname> <given-names>H.</given-names></name> <name><surname>Beygelzimer</surname> <given-names>A.</given-names></name> <name><surname>Alche&#x2019;-Buc</surname> <given-names>F. D.</given-names></name> <name><surname>Fox</surname> <given-names>E.</given-names></name> <name><surname>Garnett</surname> <given-names>R.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Curran Associates, Inc</publisher-name>), <fpage>8024</fpage>&#x2013;<lpage>8035</lpage>.</citation></ref>
<ref id="B116"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Patil</surname> <given-names>P. U.</given-names></name> <name><surname>Lande</surname> <given-names>S. B.</given-names></name> <name><surname>Nagalkar</surname> <given-names>V. J.</given-names></name> <name><surname>Nikam</surname> <given-names>S. B.</given-names></name> <name><surname>Wakchaure</surname> <given-names>G. C.</given-names></name></person-group> (<year>2021</year>). <article-title>Grading and sorting technique of dragon fruits using machine learning algorithms.</article-title> <source><italic>J. Agric. Food Res.</italic></source> <volume>4</volume>:<issue>100118</issue>. <pub-id pub-id-type="doi">10.1016/j.jafr.2021.100118</pub-id></citation></ref>
<ref id="B117"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>H.</given-names></name> <name><surname>Huang</surname> <given-names>B.</given-names></name> <name><surname>Shao</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>General improved SSD model for picking object recognition of multiple fruits in natural environment.</article-title> <source><italic>Nong Ye Gong Cheng Xue Bao</italic></source> <volume>34</volume> <fpage>155</fpage>&#x2013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.11975/j.issn.1002-6819.2018.16.020</pub-id></citation></ref>
<ref id="B118"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>P&#x00E9;rez-Borrero</surname> <given-names>I.</given-names></name> <name><surname>Mar&#x00ED;n-Santos</surname> <given-names>D.</given-names></name> <name><surname>Geg&#x00FA;ndez-Arias</surname> <given-names>M. E.</given-names></name> <name><surname>Cort&#x00E9;s-Ancos</surname> <given-names>E.</given-names></name></person-group> (<year>2020</year>). <article-title>A fast and accurate deep learning method for strawberry instance segmentation.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>178</volume>:<issue>105736</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105736</pub-id></citation></ref>
<ref id="B119"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>P&#x00E9;rez-Borrero</surname> <given-names>I.</given-names></name> <name><surname>Marin-Santos</surname> <given-names>D.</given-names></name> <name><surname>Vasallo-Vazquez</surname> <given-names>M. J.</given-names></name> <name><surname>Gegundez-Arias</surname> <given-names>M. E.</given-names></name></person-group> (<year>2021</year>). <article-title>A new deep-learning strawberry instance segmentation methodology based on a fully convolutional neural network.</article-title> <source><italic>Neural Comput. Appl.</italic></source> <volume>33</volume> <fpage>15059</fpage>&#x2013;<lpage>15071</lpage>.</citation></ref>
<ref id="B120"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pinheiro</surname> <given-names>P. O.</given-names></name> <name><surname>Collobert</surname> <given-names>R.</given-names></name> <name><surname>Dollar</surname> <given-names>P.</given-names></name></person-group> (<year>2015</year>). <article-title>Learning to segment object candidates.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. Available online at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/pdf/1506.06204.pdf">https://arxiv.org/pdf/1506.06204.pdf</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B121"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>C. R.</given-names></name> <name><surname>Su</surname> <given-names>H.</given-names></name> <name><surname>Mo</surname> <given-names>K.</given-names></name> <name><surname>Guibas</surname> <given-names>L. J.</given-names></name></person-group> (<year>2017a</year>). &#x201C;<article-title>PointNet: deep learning on point Sets for 3D classification and segmentation</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Honolulu, HI</publisher-loc>.</citation></ref>
<ref id="B122"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>C. R.</given-names></name> <name><surname>Yi</surname> <given-names>L.</given-names></name> <name><surname>Su</surname> <given-names>H.</given-names></name> <name><surname>Guibas</surname> <given-names>L. J.</given-names></name></person-group> (<year>2017b</year>). <source><italic>PointNet++: Deep Hierarchical Feature Learning on Point Sets in a Metric Space.</italic></source> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>), <fpage>5099</fpage>&#x2013;<lpage>5108</lpage>.</citation></ref>
<ref id="B123"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Quinlan</surname> <given-names>J. R.</given-names></name></person-group> (<year>1986</year>). <article-title>Induction of decision trees.</article-title> <source><italic>Mach. Learn.</italic></source> <volume>1</volume> <fpage>81</fpage>&#x2013;<lpage>106</lpage>.</citation></ref>
<ref id="B124"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Redmon</surname> <given-names>J.</given-names></name> <name><surname>Divvala</surname> <given-names>S.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Farhadi</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>You only look once: unified, real-time object detection</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Las Vegas, NV</publisher-loc>, <fpage>779</fpage>&#x2013;<lpage>788</lpage>.</citation></ref>
<ref id="B125"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Redmon</surname> <given-names>J.</given-names></name> <name><surname>Farhadi</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>YOLO9000: better, faster, stronger</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</italic></source>, <publisher-loc>Honolulu, HI</publisher-loc>, <fpage>6517</fpage>&#x2013;<lpage>6525</lpage>.</citation></ref>
<ref id="B126"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Redmon</surname> <given-names>J.</given-names></name> <name><surname>Farhadi</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>YOLOv3: an incremental improvement.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1804.02767v1">http://arxiv.org/abs/1804.02767v1</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B127"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rehman</surname> <given-names>H. U.</given-names></name> <name><surname>Miura</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Viewpoint planning for automated fruit harvesting using deep learning</article-title>,&#x201D; in <source><italic>Proceedings of the 2021 IEEE/SICE International Symposium on System Integration (SII)</italic>,</source> <publisher-loc>Iwaki</publisher-loc>, <fpage>409</fpage>&#x2013;<lpage>414</lpage>. <pub-id pub-id-type="doi">10.1109/IEEECONF49454.2021.9382628</pub-id></citation></ref>
<ref id="B128"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rehman</surname> <given-names>S. U.</given-names></name> <name><surname>Tu</surname> <given-names>S.</given-names></name> <name><surname>Waqas</surname> <given-names>M.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name> <name><surname>Rehman</surname> <given-names>O. U.</given-names></name> <name><surname>Ahmad</surname> <given-names>B.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Unsupervised pre-trained filter learning approach for efficient convolution neural network.</article-title> <source><italic>Neurocomputing</italic></source> <volume>365</volume> <fpage>171</fpage>&#x2013;<lpage>190</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2019.06.084</pub-id></citation></ref>
<ref id="B129"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Faster R-CNN: towards real-time object detection with region proposal networks.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>39</volume> <fpage>1137</fpage>&#x2013;<lpage>1149</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2016.2577031</pub-id> <pub-id pub-id-type="pmid">27295650</pub-id></citation></ref>
<ref id="B130"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Riegler</surname> <given-names>G.</given-names></name> <name><surname>Ulusoy</surname> <given-names>A. O.</given-names></name> <name><surname>Geiger</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Octnet: learning deep 3d representations at high resolutions</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 IEEE Conference on Computer Vision and Pattern Recognition</italic></source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B131"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ronneberger</surname> <given-names>O.</given-names></name> <name><surname>Fischer</surname> <given-names>P.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). &#x201C;<article-title>U-net: convolutional networks for biomedical image segmentation</article-title>,&#x201D; in <source><italic>Medical Image Computing and Computer-Assisted Intervention &#x2013; MICCAI 2015. MICCAI 2015. Lecture Notes in Computer Science</italic></source>, <volume>Vol. 9351</volume> <role>eds</role> <person-group person-group-type="editor"><name><surname>Navab</surname> <given-names>N.</given-names></name> <name><surname>Hornegger</surname> <given-names>J.</given-names></name> <name><surname>Wells</surname> <given-names>W.</given-names></name> <name><surname>Frangi</surname> <given-names>A.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>234</fpage>&#x2013;<lpage>241</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-24574-4_28</pub-id></citation></ref>
<ref id="B132"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rudolph</surname> <given-names>R.</given-names></name> <name><surname>Herzog</surname> <given-names>K.</given-names></name> <name><surname>T&#x00F6;pfer</surname> <given-names>R.</given-names></name> <name><surname>Steinhage</surname> <given-names>V.</given-names></name></person-group> (<year>2019</year>). <article-title>Efficient identification, localization and quantification of grapevine inflorescences and flowers in unprepared field images using fully convolutional networks.</article-title> <source><italic>Vitis</italic></source> <volume>58</volume> <fpage>95</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.5073/vitis.2019.58.95-104</pub-id></citation></ref>
<ref id="B133"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Russell</surname> <given-names>B. C.</given-names></name> <name><surname>Torralba</surname> <given-names>A.</given-names></name> <name><surname>Murphy</surname> <given-names>K. P.</given-names></name> <name><surname>Freeman</surname> <given-names>W. T.</given-names></name></person-group> (<year>2007</year>). <article-title>LabelMe: a database and web-based tool for image annotation.</article-title> <source><italic>Int. J. Comput. Vis.</italic></source> <volume>77</volume> <fpage>157</fpage>&#x2013;<lpage>173</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-007-0090-8</pub-id></citation></ref>
<ref id="B134"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sa</surname> <given-names>I.</given-names></name> <name><surname>Ge</surname> <given-names>Z.</given-names></name> <name><surname>Dayoub</surname> <given-names>F.</given-names></name> <name><surname>Upcroft</surname> <given-names>B.</given-names></name> <name><surname>Perez</surname> <given-names>T.</given-names></name> <name><surname>McCool</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>Deepfruits: a fruit detection system using deep neural networks.</article-title> <source><italic>Sensors</italic></source> <volume>16</volume>:<issue>1222</issue>. <pub-id pub-id-type="doi">10.3390/s16081222</pub-id> <pub-id pub-id-type="pmid">27527168</pub-id></citation></ref>
<ref id="B135"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saedi</surname> <given-names>S. I.</given-names></name> <name><surname>Khosravi</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>A deep neural network approach towards real-time on-branch fruit recognition for precision horticulture.</article-title> <source><italic>Expert Syst. Appl.</italic></source> <volume>159</volume>:<issue>113594</issue>. <pub-id pub-id-type="doi">10.1016/j.eswa.2020.113594</pub-id></citation></ref>
<ref id="B136"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Santos</surname> <given-names>T. T.</given-names></name> <name><surname>de Souza</surname> <given-names>L. L.</given-names></name> <name><surname>dos Santos</surname> <given-names>A. A.</given-names></name> <name><surname>Avila</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Grape detection, segmentation, and tracking using deep neural networks and three-dimensional association.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>170</volume>:<issue>105247</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105247</pub-id></citation></ref>
<ref id="B137"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schuster</surname> <given-names>M.</given-names></name> <name><surname>Paliwal</surname> <given-names>K. K.</given-names></name></person-group> (<year>1997</year>). <article-title>Bidirectional recurrent neural networks.</article-title> <source><italic>IEEE Trans. Signal Process.</italic></source> <volume>45</volume> <fpage>2673</fpage>&#x2013;<lpage>2681</lpage>.</citation></ref>
<ref id="B138"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shelhamer</surname> <given-names>E.</given-names></name> <name><surname>Long</surname> <given-names>J.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name></person-group> (<year>2017</year>). <article-title>Fully convolutional networks for semantic segmentation.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>39</volume> <fpage>640</fpage>&#x2013;<lpage>651</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2016.2572683</pub-id> <pub-id pub-id-type="pmid">27244717</pub-id></citation></ref>
<ref id="B139"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Silwal</surname> <given-names>A.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name></person-group> (<year>2016</year>). <article-title>A hierarchical approach to apple identification for robotic harvesting.</article-title> <source><italic>Trans. ASABE</italic></source> <volume>59</volume> <fpage>1079</fpage>&#x2013;<lpage>1086</lpage>. <pub-id pub-id-type="doi">10.13031/trans.59.11619</pub-id></citation></ref>
<ref id="B140"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>Very deep convolutional networks for large-scale image recognition.</article-title> <source><italic>Comput. Sci.</italic></source></citation></ref>
<ref id="B141"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>Z.</given-names></name> <name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Cui</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Kiwifruit detection in field images using faster R-CNN with VGG16.</article-title> <source><italic>IFAC Papersonline</italic></source> <volume>52</volume> <fpage>76</fpage>&#x2013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.1016/j.ifacol.2019.12.500</pub-id></citation></ref>
<ref id="B142"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stein</surname> <given-names>M.</given-names></name> <name><surname>Bargoti</surname> <given-names>S.</given-names></name> <name><surname>Underwood</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Image based mango fruit detection, localisation and yield estimation using multiple view geometry.</article-title> <source><italic>Sensors</italic></source> <volume>16</volume>:<issue>1915</issue>. <pub-id pub-id-type="doi">10.3390/s16111915</pub-id> <pub-id pub-id-type="pmid">27854271</pub-id></citation></ref>
<ref id="B143"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Ge</surname> <given-names>X.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>Detection of key organs in tomato based on deep migration learning in a complex background.</article-title> <source><italic>Agriculture</italic></source> <volume>8</volume>:<issue>196</issue>. <pub-id pub-id-type="doi">10.3390/agriculture8120196</pub-id></citation></ref>
<ref id="B144"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Wu</surname> <given-names>M.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>Detection of tomato organs based on convolutional neural network under the overlap and occlusion backgrounds.</article-title> <source><italic>Mach. Vis. Appl.</italic></source> <volume>31</volume>:<issue>31</issue>. <pub-id pub-id-type="doi">10.1007/s00138-020-01081-6</pub-id></citation></ref>
<ref id="B145"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Supper</surname> <given-names>J.</given-names></name> <name><surname>Spieth</surname> <given-names>C.</given-names></name> <name><surname>Zell</surname> <given-names>A.</given-names></name></person-group> (<year>2007</year>). &#x201C;<article-title>Reconstructing linear gene regulatory networks</article-title>,&#x201D; in <source><italic>Evolutionary Computation, Machine Learning and Data Mining in Bioinformatics</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Marchiori</surname> <given-names>E.</given-names></name> <name><surname>Moore</surname> <given-names>J. H.</given-names></name> <name><surname>Rajapakse</surname> <given-names>J. C.</given-names></name></person-group> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>270</fpage>&#x2013;<lpage>279</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-540-71783-6_26</pub-id></citation></ref>
<ref id="B146"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Sermanet</surname> <given-names>P.</given-names></name> <name><surname>Rabinovich</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). &#x201C;<article-title>Going deeper with convolutions</article-title>,&#x201D; in <source><italic>Proceedings of the 2015 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Boston, MA</publisher-loc>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298594</pub-id></citation></ref>
<ref id="B147"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Vanhoucke</surname> <given-names>V.</given-names></name> <name><surname>Ioffe</surname> <given-names>S.</given-names></name> <name><surname>Shlens</surname> <given-names>J.</given-names></name> <name><surname>Wojna</surname> <given-names>Z.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Rethinking the inception architecture for computer vision</article-title>,&#x201D; in <source><italic>Proceedings of the 2016 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, <publisher-loc>Las Vegas, NV</publisher-loc>, <fpage>2818</fpage>&#x2013;<lpage>2826</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.308</pub-id></citation></ref>
<ref id="B148"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>W. S.</given-names></name> <name><surname>Gan</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name></person-group> (<year>2018</year>). <article-title>Recognising blueberry fruit of different maturity using histogram oriented gradients and colour features in outdoor scenes.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>176</volume> <fpage>59</fpage>&#x2013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2018.08.011</pub-id></citation></ref>
<ref id="B149"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>C.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <source><italic>Look Closer to Segment Better: Boundary Patch Refinement for Instance Segmentation.</italic></source> Available online at: <ext-link ext-link-type="uri" xlink:href="https://docplayer.net/amp/218770859-Look-closer-to-segment-better-boundary-patch-refinement-for-instance-segmentation-supplementary-material.html">https://docplayer.net/amp/218770859-Look-closer-to-segment-better-boundary-patch-refinement-for-instance-segmentation-supplementary-material.html</ext-link> <comment>(accessed September 2021)</comment>.</citation></ref>
<ref id="B150"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thendral</surname> <given-names>R.</given-names></name> <name><surname>Suhasini</surname> <given-names>A.</given-names></name> <name><surname>Senthil</surname> <given-names>N.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>A comparative analysis of edge and color based segmentation for orange fruit recognition</article-title>,&#x201D; in <source><italic>Proceedings of the 2014 International Conference on Communication and Signal Processing (ICCSP)</italic></source>, <publisher-loc>Melmaruvathur</publisher-loc>. <pub-id pub-id-type="doi">10.1109/ICCSP.2014.6949884</pub-id></citation></ref>
<ref id="B151"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>E.</given-names></name> <name><surname>Liang</surname> <given-names>Z.</given-names></name></person-group> (<year>2020</year>). <article-title>Instance segmentation of apple flowers using the improved mask R-CNN model</article-title>. <source><italic>Biosyst. Eng.</italic></source> <volume>193</volume>, <fpage>264</fpage>&#x2013;<lpage>278</lpage>.</citation></ref>
<ref id="B152"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>E.</given-names></name> <name><surname>Liang</surname> <given-names>Z.</given-names></name></person-group> (<year>2019</year>). <article-title>Apple detection during different growth stages in orchards using the improved YOLO-V3 model.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>157</volume> <fpage>417</fpage>&#x2013;<lpage>426</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.01.012</pub-id></citation></ref>
<ref id="B153"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Triggs</surname> <given-names>B.</given-names></name> <name><surname>McLauchlan</surname> <given-names>P. F.</given-names></name> <name><surname>Hartley</surname> <given-names>R. I.</given-names></name> <name><surname>Fitzgibbon</surname> <given-names>A. W.</given-names></name></person-group> (<year>2002</year>). &#x201C;<article-title>Bundle adjustment &#x2014; A modern synthesis</article-title>,&#x201D; in <source><italic>Vision Algorithms: Theory and Practice. IWVA 1999. Lecture Notes in Computer Science</italic></source>, <volume>Vol. 1883</volume> <role>eds</role> <person-group person-group-type="editor"><name><surname>Triggs</surname> <given-names>B.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name> <name><surname>Szeliski</surname> <given-names>R.</given-names></name></person-group> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>298</fpage>&#x2013;<lpage>372</lpage>. <pub-id pub-id-type="doi">10.1007/3-540-44480-7_21</pub-id></citation></ref>
<ref id="B154"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tsai</surname> <given-names>C.</given-names></name> <name><surname>Chiu</surname> <given-names>C.</given-names></name></person-group> (<year>2008</year>). <article-title>Developing a feature weight self-adjustment mechanism for a K-means clustering algorithm.</article-title> <source><italic>Comput. Stat. Data Anal.</italic></source> <volume>52</volume> <fpage>4658</fpage>&#x2013;<lpage>4672</lpage>. <pub-id pub-id-type="doi">10.1016/j.csda.2008.03.002</pub-id></citation></ref>
<ref id="B155"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tsoulias</surname> <given-names>N.</given-names></name> <name><surname>Paraforos</surname> <given-names>D. S.</given-names></name> <name><surname>Xanthopoulos</surname> <given-names>G.</given-names></name> <name><surname>Zude-Sasse</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Apple shape detection based on geometric and radiometric features using a LiDAR laser scanner.</article-title> <source><italic>Remote Sens.</italic></source> <volume>12</volume>:<issue>2481</issue>. <pub-id pub-id-type="doi">10.3390/rs12152481</pub-id></citation></ref>
<ref id="B156"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>S.</given-names></name> <name><surname>Pang</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Zhuang</surname> <given-names>N.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Zheng</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Passion fruit detection and counting based on multiple scale faster R-CNN using RGB-D images.</article-title> <source><italic>Precis. Agric.</italic></source> <volume>21</volume> <fpage>1072</fpage>&#x2013;<lpage>1091</lpage>. <pub-id pub-id-type="doi">10.1007/s11119-020-09709-3</pub-id></citation></ref>
<ref id="B157"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>S.</given-names></name> <name><surname>Xue</surname> <given-names>Y.</given-names></name> <name><surname>Zheng</surname> <given-names>C.</given-names></name> <name><surname>Qi</surname> <given-names>Y.</given-names></name> <name><surname>Wan</surname> <given-names>H.</given-names></name> <name><surname>Mao</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Detection of passion fruits and maturity classification using red-green-blue depth images.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>175</volume> <fpage>156</fpage>&#x2013;<lpage>167</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2018.09.004</pub-id></citation></ref>
<ref id="B158"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uijlings</surname> <given-names>J. R.</given-names></name> <name><surname>van de Sande</surname> <given-names>K. E. A.</given-names></name> <name><surname>Gevers</surname> <given-names>T.</given-names></name> <name><surname>Smeulders</surname> <given-names>A. W. M.</given-names></name></person-group> (<year>2013</year>). <article-title>Selective search for object recognition.</article-title> <source><italic>Int. J. Comput. Vis.</italic></source> <volume>104</volume> <fpage>154</fpage>&#x2013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-013-0620-5</pub-id></citation></ref>
<ref id="B159"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vasconez</surname> <given-names>J. P.</given-names></name> <name><surname>Delpiano</surname> <given-names>J.</given-names></name> <name><surname>Vougioukas</surname> <given-names>S.</given-names></name> <name><surname>AuatCheein</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>Comparison of convolutional neural networks in fruit detection and counting: a comprehensive evaluation.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>173</volume>:<issue>105348</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105348</pub-id></citation></ref>
<ref id="B160"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wan</surname> <given-names>S.</given-names></name> <name><surname>Goudos</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Faster R-CNN for multi-class fruit detection using a robotic vision system.</article-title> <source><italic>Comput. Netw.</italic></source> <volume>168</volume>:<issue>107036</issue>. <pub-id pub-id-type="doi">10.1016/j.comnet.2019.107036</pub-id></citation></ref>
<ref id="B161"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Dong</surname> <given-names>L.</given-names></name> <name><surname>Zhou</surname> <given-names>H.</given-names></name> <name><surname>Luo</surname> <given-names>L.</given-names></name> <name><surname>Lin</surname> <given-names>G.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>YOLOv3-litchi detection method of densely distributed litchi in large vision scenes.</article-title> <source><italic>Math. Probl. Eng.</italic></source> <volume>2021</volume>:<issue>8883015</issue>. <pub-id pub-id-type="doi">10.1155/2021/8883015</pub-id></citation></ref>
<ref id="B162"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Xu</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Non-destructive detection of apple multi-quality parameters based on hyperspectral imaging technology and 3D-CNN.</article-title> <source><italic>Nanjing NongyeDaxueXuebao</italic></source> <volume>43</volume> <fpage>178</fpage>&#x2013;<lpage>185</lpage>. <pub-id pub-id-type="doi">10.7685/jnau.201906067</pub-id></citation></ref>
<ref id="B163"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Zeng</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Zou</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>Extraction of litchi fruit pericarp defect based on a fully convolutional neural network.</article-title> <source><italic>Hua Nan Nong Ye Da XueXue Bao</italic></source> <volume>39</volume> <fpage>104</fpage>&#x2013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.7671/j.issn.1001-411X.2018.06.016</pub-id></citation></ref>
<ref id="B164"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Whitty</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>DeepPhenology: estimation of apple flower phenology distributions based on deep learning.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>185</volume>:<issue>106123</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2021.106123</pub-id></citation></ref>
<ref id="B165"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>C.</given-names></name> <name><surname>Han</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>Counting method of cherry tomato fruits in greenhouses based on deep learning.</article-title> <source><italic>J. China Univ. Metrol.</italic></source> <volume>32</volume> <fpage>93</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.2096-2835.2021.01.013</pub-id></citation></ref>
<ref id="B166"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wittstruck</surname> <given-names>L.</given-names></name> <name><surname>K&#x00FC;hling</surname> <given-names>I.</given-names></name> <name><surname>Trautz</surname> <given-names>D.</given-names></name> <name><surname>Kohlbrecher</surname> <given-names>M.</given-names></name> <name><surname>Jarmer</surname> <given-names>T.</given-names></name></person-group> (<year>2021</year>). <article-title>UAV-based RGB imagery for Hokkaido pumpkin (<italic>Cucurbita max</italic>.) detection and yield estimation.</article-title> <source><italic>Sensors</italic></source> <volume>21</volume>:<issue>118</issue>. <pub-id pub-id-type="doi">10.3390/s21010118</pub-id> <pub-id pub-id-type="pmid">33375474</pub-id></citation></ref>
<ref id="B167"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wouters</surname> <given-names>N.</given-names></name> <name><surname>De Ketelaere</surname> <given-names>B.</given-names></name> <name><surname>De Baerdemaeker</surname> <given-names>J.</given-names></name> <name><surname>Saeys</surname> <given-names>W.</given-names></name></person-group> (<year>2012</year>). <article-title>Hyperspectral waveband selection for automatic detection of floral pear buds.</article-title> <source><italic>Precis. Agric.</italic></source> <volume>14</volume> <fpage>86</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1007/s11119-012-9279-0</pub-id></citation></ref>
<ref id="B168"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>A.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Ren</surname> <given-names>T.</given-names></name></person-group> (<year>2020</year>). <article-title>Detection of apple defect using laser-induced light backscattering imaging and convolutional neural network.</article-title> <source><italic>Comput. Electr. Eng.</italic></source> <volume>81</volume>:<issue>106454</issue>. <pub-id pub-id-type="doi">10.1016/j.compeleceng.2019.106454</pub-id></citation></ref>
<ref id="B169"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>D.</given-names></name> <name><surname>Lv</surname> <given-names>S.</given-names></name> <name><surname>Jiang</surname> <given-names>M.</given-names></name> <name><surname>Song</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Using channel pruning-based YOLO v4 deep learning algorithm for the real-time and accurate detection of apple flowers in natural environments.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>178</volume>:<issue>105742</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105742</pub-id></citation></ref>
<ref id="B170"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>S.</given-names></name> <name><surname>Tung</surname> <given-names>H.</given-names></name> <name><surname>Hsu</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>Deep learning for automatic quality grading of mangoes: methods and insights</article-title>,&#x201D; in <source><italic>Proceedings of the 2020 19th IEEE International Conference on Machine Learning and Applications (ICMLA)</italic></source>, <publisher-loc>Miami, FL</publisher-loc>, <fpage>446</fpage>&#x2013;<lpage>453</lpage>. <pub-id pub-id-type="doi">10.1109/ICMLA51294.2020.00076</pub-id></citation></ref>
<ref id="B171"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>R.</given-names></name> <name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <article-title>Unmanned aerial vehicle vision detection technology of green mango on tree in natural environment</article-title>. <source><italic>Trans. Chin. Soc. Agric. Mach.</italic></source> <volume>49</volume>, <fpage>23</fpage>&#x2013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2018.11.003</pub-id></citation></ref>
<ref id="B172"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Liao</surname> <given-names>H.</given-names></name> <name><surname>Kang</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>Overlapping citrus segmentation and reconstruction based on mask R-CNN model and concave region simplification and distance analysis</article-title>. <source><italic>J. Phys. Conf. Ser.</italic></source> <volume>1345</volume>:<issue>32064</issue>. <pub-id pub-id-type="doi">10.1088/1742-6596/1345/3/032064</pub-id></citation></ref>
<ref id="B173"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Zhong</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Zheng</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). <article-title>Litchi flower and leaf segmentation and recognition based on deep semantic segmentation.</article-title> <source><italic>Trans. Chin. Soc. Agric. Machinery</italic></source> <volume>52</volume> <fpage>252</fpage>&#x2013;<lpage>258</lpage>.</citation></ref>
<ref id="B175"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>Y.</given-names></name> <name><surname>Ge</surname> <given-names>Y.</given-names></name> <name><surname>From</surname> <given-names>P. J.</given-names></name></person-group> (<year>2020</year>). <article-title>An obstacle separation method for robotic picking of fruits in clusters.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>175</volume>:<issue>105397</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105397</pub-id></citation></ref>
<ref id="B176"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Ding</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Detection of small fruit target based on improved DenseNet.</article-title> <source><italic>J. Zhejiang Univ.</italic></source> <volume>55</volume> <fpage>377</fpage>&#x2013;<lpage>385</lpage>. <pub-id pub-id-type="doi">10.3785/j.issn.1008-973X.2021.02.018</pub-id></citation></ref>
<ref id="B177"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>W.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Nondestructive detection of yellow peach quality parameters based on 3D-CNN and hyperspectral images.</article-title> <source><italic>J. Phys. Conf. Ser.</italic></source> <volume>1682</volume>:<issue>012030</issue>. <pub-id pub-id-type="doi">10.1088/1742-6596/1682/1/012030</pub-id></citation></ref>
<ref id="B178"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Su</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>F.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Recognition of <italic>Rosa roxbunghii</italic> in natural environment based on improved faster RCNN.</article-title> <source><italic>Nong Ye Gong Cheng Xue Bao</italic></source> <volume>35</volume> <fpage>143</fpage>&#x2013;<lpage>150</lpage>. <pub-id pub-id-type="doi">10.11975/j.issn.1002-6819.2019.18.018</pub-id></citation></ref>
<ref id="B179"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Xiong</surname> <given-names>L.</given-names></name> <name><surname>Kang</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>W.</given-names></name></person-group> (<year>2019</year>). <article-title>Identification and reconstruction of citrus branches under complex background based on mask R-CNN.</article-title> <source><italic>Trans. Chin. Soc. Agric. Machinery</italic></source> <volume>50</volume> <fpage>22</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2019.08.003</pub-id></citation></ref>
<ref id="B180"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>C. H.</given-names></name> <name><surname>Xiong</surname> <given-names>L. Y.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Shi</surname> <given-names>G.</given-names></name> <name><surname>Kuremot</surname> <given-names>T.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Integrated detection of citrus fruits and branches using a convolutional neural network.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>174</volume>:<issue>105469</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105469</pub-id></citation></ref>
<ref id="B181"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>W.</given-names></name> <name><surname>Wen</surname> <given-names>H.</given-names></name> <name><surname>Ning</surname> <given-names>Z.</given-names></name> <name><surname>Ye</surname> <given-names>J.</given-names></name> <name><surname>Dong</surname> <given-names>Z.</given-names></name> <name><surname>Luo</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>Fruit detection and pose estimation for grape Cluster&#x2013;Harvesting robot using binocular imagery based on deep neural networks.</article-title> <source><italic>Front. Robot. AI</italic></source> <volume>8</volume>:<issue>626989</issue>. <pub-id pub-id-type="doi">10.3389/frobt.2021.626989</pub-id> <pub-id pub-id-type="pmid">34239899</pub-id></citation></ref>
<ref id="B182"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>G.</given-names></name> <name><surname>Ma</surname> <given-names>B.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Nondestructive identification of pesticide residues on the hami melon surface using deep feature fusion by Vis/NIR spectroscopy and 1D-CNN.</article-title> <source><italic>J. Food Process Eng.</italic></source> <volume>44</volume>:<issue>e13602</issue>. <pub-id pub-id-type="doi">10.1111/jfpe.13602</pub-id></citation></ref>
<ref id="B183"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>X.</given-names></name> <name><surname>Lu</surname> <given-names>H.</given-names></name> <name><surname>Wu</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>Development of deep learning method for predicting firmness and soluble solid content of postharvest korla fragrant pear using Vis/NIR hyperspectral reflectance imaging.</article-title> <source><italic>Postharvest Biol. Technol.</italic></source> <volume>141</volume> <fpage>39</fpage>&#x2013;<lpage>49</lpage>. <pub-id pub-id-type="doi">10.1016/j.postharvbio.2018.02.013</pub-id></citation></ref>
<ref id="B184"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>K.</given-names></name> <name><surname>Yang</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>Fruit detection for strawberry harvesting robot in non-structural environment based on mask-RCNN.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>163</volume>:<issue>104846</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.06.001</pub-id></citation></ref>
<ref id="B185"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yuan</surname> <given-names>B.</given-names></name> <name><surname>Zhan</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>C.</given-names></name></person-group> (<year>2017</year>). <article-title>Evolution of a development model for fruit industry against background of an aging population: intensive or extensive adjustment.</article-title> <source><italic>Sustainability</italic></source> <volume>10</volume>:<issue>49</issue>. <pub-id pub-id-type="doi">10.3390/su10010049</pub-id></citation></ref>
<ref id="B186"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yue</surname> <given-names>X.-Q.</given-names></name> <name><surname>Shang</surname> <given-names>Z.-Y.</given-names></name> <name><surname>Yang</surname> <given-names>J.-Y.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>Y.-Q.</given-names></name></person-group> (<year>2020</year>). <article-title>A smart data-driven rapid method to recognize the strawberry maturity.</article-title> <source><italic>Inf. Process. Agric.</italic></source> <volume>7</volume> <fpage>575</fpage>&#x2013;<lpage>584</lpage>. <pub-id pub-id-type="doi">10.1016/j.inpa.2019.10.005</pub-id></citation></ref>
<ref id="B187"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zabawa</surname> <given-names>L.</given-names></name> <name><surname>Kicherer</surname> <given-names>A.</given-names></name> <name><surname>Klingbeil</surname> <given-names>L.</given-names></name> <name><surname>T&#x00F6;pfer</surname> <given-names>R.</given-names></name> <name><surname>Kuhlmann</surname> <given-names>H.</given-names></name> <name><surname>Roscher</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>Counting of grapevine berries in images via semantic segmentation using convolutional neural networks.</article-title> <source><italic>ISPRS J. Photogramm. Remote Sens.</italic></source> <volume>164</volume> <fpage>73</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2020.04.002</pub-id></citation></ref>
<ref id="B188"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>T.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Ma</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Luo</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2019</year>). <article-title>Localization and defect detection of jujubes based on search of shortest path between frames and ensemble-CNN model.</article-title> <source><italic>Trans. Chin. Soc. Agric. Machinery</italic></source> <volume>50</volume> <fpage>307</fpage>&#x2013;<lpage>314</lpage>. <pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2019.02.035</pub-id></citation></ref>
<ref id="B189"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>L.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Gao</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <article-title>Branch detection for apple trees trained in fruiting wall architecture using depth features and regions-convolutional neural network (R-CNN).</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>155</volume> <fpage>386</fpage>&#x2013;<lpage>393</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2018.10.029</pub-id></citation></ref>
<ref id="B190"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Karkee</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Yaqoob</surname> <given-names>M.</given-names></name> <name><surname>Fu</surname> <given-names>L.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Multi-class object detection using faster R-CNN and estimation of shaking locations for automated shake-and-catch apple harvesting.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>173</volume>:<issue>105384</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105384</pub-id></citation></ref>
<ref id="B191"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Jia</surname> <given-names>J.</given-names></name> <name><surname>Gui</surname> <given-names>G.</given-names></name> <name><surname>Hao</surname> <given-names>X.</given-names></name> <name><surname>Gao</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep learning based improved classification system for designing tomato harvesting robot.</article-title> <source><italic>IEEE Access</italic></source> <volume>6</volume> <fpage>67940</fpage>&#x2013;<lpage>67950</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2018.2879324</pub-id></citation></ref>
<ref id="B192"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Q.</given-names></name> <name><surname>Gao</surname> <given-names>G.</given-names></name></person-group> (<year>2020</year>). <article-title>Prioritizing robotic grasping of stacked fruit clusters based on stalk location in RGB-D images.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>172</volume>:<issue>105359</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105359</pub-id></citation></ref>
<ref id="B194"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Q.</given-names></name> <name><surname>Kong</surname> <given-names>P.</given-names></name> <name><surname>Min</surname> <given-names>J.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Liang</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>A review of deep learning methods for the detection and classification of pulmonary nodules.</article-title> <source><italic>Sheng Wu Yi Xue Gong Cheng Xue Za Zhi</italic></source> <volume>36</volume> <fpage>1060</fpage>&#x2013;<lpage>1068</lpage>. <pub-id pub-id-type="doi">10.7507/1001-5515.201903027</pub-id> <pub-id pub-id-type="pmid">31875384</pub-id></citation></ref>
<ref id="B195"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Gong</surname> <given-names>L.</given-names></name> <name><surname>Zhou</surname> <given-names>B.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>Detecting tomatoes in greenhouse scenes by combining AdaBoost classifier and colour analysis.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>148</volume> <fpage>127</fpage>&#x2013;<lpage>137</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2016.05.001</pub-id></citation></ref>
<ref id="B196"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Z.</given-names></name> <name><surname>Zheng</surname> <given-names>P.</given-names></name> <name><surname>Xu</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>Object detection with deep learning: a review.</article-title> <source><italic>IEEE Trans. Neural Netw. Learn. Syst.</italic></source> <volume>30</volume> <fpage>3212</fpage>&#x2013;<lpage>3232</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2018.2876865</pub-id> <pub-id pub-id-type="pmid">30703038</pub-id></citation></ref>
<ref id="B197"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>Z.</given-names></name> <name><surname>Zheng</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Unlabeled samples generated by GAN improve the person re-identification baseline in vitro</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 IEEE International Conference on Computer Vision (ICCV2017)</italic></source>, <publisher-loc>Venice</publisher-loc>, <fpage>3774</fpage>&#x2013;<lpage>3782</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.405</pub-id></citation></ref>
<ref id="B198"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>R.</given-names></name> <name><surname>Damerow</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Blanke</surname> <given-names>M. M.</given-names></name></person-group> (<year>2012</year>). <article-title>Using colour features of cv. &#x2018;Gala&#x2019; apple fruits in an orchard in image processing to predict yield.</article-title> <source><italic>Precis. Agric.</italic></source> <volume>13</volume> <fpage>568</fpage>&#x2013;<lpage>580</lpage>. <pub-id pub-id-type="doi">10.1007/s11119-012-9269-2</pub-id></citation></ref>
<ref id="B199"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>X.</given-names></name> <name><surname>Lee</surname> <given-names>W. S.</given-names></name> <name><surname>Ampatzidis</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Peres</surname> <given-names>N.</given-names></name> <name><surname>Fraisse</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Strawberry maturity classification from UAV and near-ground imaging using deep learning.</article-title> <source><italic>Smart Agric. Technol.</italic></source> <volume>1</volume>:<issue>100001</issue>. <pub-id pub-id-type="doi">10.1016/j.atech.2021.100001</pub-id></citation></ref>
<ref id="B200"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z.</given-names></name> <name><surname>Song</surname> <given-names>Z.</given-names></name> <name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Gao</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Cui</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Real-time kiwifruit detection in orchard using deep learning on android&#x2122; smartphones for yield estimation.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>179</volume>:<issue>105856</issue>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105856</pub-id></citation></ref>
<ref id="B201"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Park</surname> <given-names>T.</given-names></name> <name><surname>Isola</surname> <given-names>P.</given-names></name> <name><surname>Efros</surname> <given-names>A. A.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Unpaired image-to-image translation using cycle-consistent adversarial networks</article-title>,&#x201D; in <source><italic>Proceedings of the 2017 IEEE International Conference on Computer Vision (ICCV)</italic></source>, <publisher-loc>Venice</publisher-loc>, <fpage>2242</fpage>&#x2013;<lpage>2251</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.244</pub-id></citation></ref>
<ref id="B202"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Sharma</surname> <given-names>A. S.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Jiao</surname> <given-names>T.</given-names></name> <name><surname>Ouyang</surname> <given-names>Q.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Rapid on-site identification of pesticide residues in tea by one-dimensional convolutional neural network coupled with surface-enhanced Raman scattering.</article-title> <source><italic>Spectrochim. Acta A Mol. Biomol. Spectrosc.</italic></source> <volume>246</volume>:<issue>118994</issue>. <pub-id pub-id-type="doi">10.1016/j.saa.2020.118994</pub-id> <pub-id pub-id-type="pmid">33038862</pub-id></citation></ref>
<ref id="B203"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>N.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Hu</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Tan</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Deep learning for smart agriculture: concepts, tools, applications, and opportunities.</article-title> <source><italic>Int. J. Agric. Biol. Eng.</italic></source> <volume>11</volume> <fpage>32</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.25165/ijabe.v11i4.4475</pub-id></citation></ref>
</ref-list>
<glossary>
<title>Abbreviations</title>
<def-list id="DL1">
<def-item><term>CNN</term><def><p>convolutional neural network</p></def></def-item>
<def-item><term>DBN</term><def><p>deep belief network</p></def></def-item>
<def-item><term>RNN</term><def><p>recurrent neural network</p></def></def-item>
<def-item><term>VGG</term><def><p>visual geometry group</p></def></def-item>
<def-item><term>DTI</term><def><p>decision tree induction</p></def></def-item>
<def-item><term>SVM</term><def><p>support vector machine</p></def></def-item>
<def-item><term>2D</term><def><p>two-dimensional</p></def></def-item>
<def-item><term>ms-MLP</term><def><p>multiscale-multilayered perceptron</p></def></def-item>
<def-item><term>HoG</term><def><p>histogram of oriented gradient</p></def></def-item>
<def-item><term>ML</term><def><p>machine learning</p></def></def-item>
<def-item><term>GLCM</term><def><p>gray-level co-occurrence matrix</p></def></def-item>
<def-item><term>CIELab</term><def><p>Commission Internationale de l&#x2019;Eclairage Laboratory</p></def></def-item>
<def-item><term>CHT</term><def><p>circular Hough transform</p></def></def-item>
<def-item><term>SLIC</term><def><p>simple linear iterative clustering</p></def></def-item>
<def-item><term>YOLO</term><def><p>you only look once</p></def></def-item>
<def-item><term>SSD</term><def><p>single shot multibox detector</p></def></def-item>
<def-item><term>mAP</term><def><p>mean average precision</p></def></def-item>
<def-item><term>STN</term><def><p>Special Transform Network</p></def></def-item>
<def-item><term>CCD</term><def><p>charge coupled device</p></def></def-item>
<def-item><term>SMOTE</term><def><p>synthetic minority oversampling technique</p></def></def-item>
<def-item><term>DC-GAN</term><def><p>deep convolutional generative adversarial network</p></def></def-item>
<def-item><term>CycleGAN</term><def><p>cycle generative adversarial network</p></def></def-item>
<def-item><term>CVAE-GAN</term><def><p>conditional autoencoder generative adversarial network</p></def></def-item>
<def-item><term>GAN</term><def><p>generative adversarial network</p></def></def-item>
<def-item><term>CPU</term><def><p>central processing unit</p></def></def-item>
<def-item><term>GPU</term><def><p>graphics processing unit</p></def></def-item>
<def-item><term>TP</term><def><p>true positive</p></def></def-item>
<def-item><term>FN</term><def><p>false negative</p></def></def-item>
<def-item><term>FP</term><def><p>false positive</p></def></def-item>
<def-item><term>TN</term><def><p>true negative</p></def></def-item>
<def-item><term>MAE</term><def><p>mean absolute error</p></def></def-item>
<def-item><term>MSE</term><def><p>mean square error</p></def></def-item>
<def-item><term>RMSE</term><def><p>root mean square error</p></def></def-item>
<def-item><term>FCN</term><def><p>full convolutional net</p></def></def-item>
<def-item><term>AV</term><def><p>unmanned aerial vehicle</p></def></def-item>
<def-item><term>MS-FRCNN</term><def><p>multiple scale Faster R-CNN</p></def></def-item>
<def-item><term>MIoU</term><def><p>mean intersection over union</p></def></def-item>
<def-item><term>SFM</term><def><p>structure from motion</p></def></def-item>
<def-item><term>ROI</term><def><p>region of interest</p></def></def-item>
<def-item><term>E-CNN</term><def><p>ensemble-convolutional neural net</p></def></def-item>
<def-item><term>NIR</term><def><p>near infrared.</p></def></def-item>
</def-list>
</glossary>
</back>
</article>
