<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1636727</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Research on persimmon fruit diameter accurate detection method based on improved RCNN instance segmentation algorithm</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Fang</surname>
<given-names>Yuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Yangyang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2644560/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Feng</surname>
<given-names>Ya</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Yougen</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2674650/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Jiang</surname>
<given-names>Haikun</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3030065/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Mechanical Engineering, Anhui University of Technology</institution>, <addr-line>Ma&#x2019;anshan</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Horticulture College, Anhui Agricultural University</institution>, <addr-line>Hefei</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>College of Information Engineering, Shaoxing Vocational &amp; Technical College</institution>, <addr-line>Shaoxing</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Institute of Vegetables, Anhui Academy of Agricultural Sciences</institution>, <addr-line>Hefei</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Zhenghong Yu, Guangdong Polytechnic of Science and Technology, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Yuanhong Li, South China Agricultural University, China</p>
<p>Dianbin Su, Shandong University of Technology, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yangyang Liu, <email xlink:href="mailto:gwglyy@163.com">gwglyy@163.com</email>; Haikun Jiang, <email xlink:href="mailto:jhk211@163.com">jhk211@163.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1636727</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Fang, Liu, Feng, Chen and Jiang</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Fang, Liu, Feng, Chen and Jiang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Aiming at the problem of inaccurate fruit recognition and fruit diameter detection in the persimmon inspection process, this research proposes a novel persimmon accurate recognition and fruit diameter detection algorithm based on the Region-based Convolutional Neural Network (RCNN) Mask and instance segmentation algorithm. The algorithm strategically targets the object of interest by integrating cropping, morphological processing, and concave point segmentation modules into the fully connected layer following the Region of Interest (RoI) feature. Initially, the algorithm separates the front and back background of the cropped target object using morphological processing to obtain a binarized image. Subsequently, concave point segmentation is applied to address sticking issues arising from overlapping or occlusion between fruits, while a template matching algorithm helps in image recognition. The improved instance segmentation algorithm enhances the segmentation accuracy of the target fruit and reduces the relative error in the fruit diameter measurement caused by sticking problems during occlusion and overlap. Notably, compared with the original algorithm, the improved Mask RCNN instance segmentation algorithm achieves a mean Average Precision (mAP) of 94.25%, representing an improvement of 8.05%, with the Mean Intersection-over-Union (MIoU) value increasing by 18.5%. The maximum relative error in fruit diameter measurement is reduced to 1.3%, while the maximum relative error in fruit thickness measurement is 1.98%, meeting the stringent requirements of orchard inspection. Overall, the proposed method enhances the precision and accuracy of fruit diameter detection, offering valuable theoretical and technical insights for intelligent inspection, yield estimation, fruit detection, and mechanized picking in the agricultural domain.</p>
</abstract>
<kwd-group>
<kwd>persimmon recognition</kwd>
<kwd>fruit diameter detection</kwd>
<kwd>Mask RCNN</kwd>
<kwd>instance segmentation algorithm</kwd>
<kwd>binarization</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<counts>
<fig-count count="12"/>
<table-count count="2"/>
<equation-count count="26"/>
<ref-count count="29"/>
<page-count count="15"/>
<word-count count="8087"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Technical Advances in Plant Science</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>An orchard is a compound ecosystem, playing a vital role in the development of the rural economy in China. The digitalization, informatization, and intelligent management of orchards are important foundation for the development of such modern ecosystems (<xref ref-type="bibr" rid="B4">Faria et&#xa0;al., 2025</xref>; <xref ref-type="bibr" rid="B18">Wang, 2025</xref>; <xref ref-type="bibr" rid="B27">Zhenyu et&#xa0;al., 2025</xref>). Moreover, China is one of the main persimmon-producing countries in the world, and the efficient identification, precise positioning, and accurate detection of the fruit diameter of persimmons is essential to realize the intelligent orchard (<xref ref-type="bibr" rid="B25">Zhang et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B13">Mao et&#xa0;al., 2025</xref>). In a complex natural environment, persimmons face serious problems, such as mutual shading and sticking, restricting the intelligent development of the orchard.</p>
<p>With the innovative development of Deep Learning (DL) recognition and detection technology in the agriculture domain, effective technical support is provided for the intelligent development of orchards (<xref ref-type="bibr" rid="B1">Bai et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B5">Fu et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B9">Ling et&#xa0;al., 2024</xref>). Therefore, local and international researchers have made some progress in research on fruit recognition and classification in complex orchard environments, utilizing neural network models, such as Faster Region-based Convolutional Neural Network (RCNN) and You Only Look Once (YOLO) algorithms, for fruit target detection in highly complex scenes (<xref ref-type="bibr" rid="B16">Tian et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B6">Hou et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B17">Ullah et&#xa0;al., 2024</xref>).</p>
<p>Xu et&#xa0;al (<xref ref-type="bibr" rid="B20">Xu et&#xa0;al., 2022</xref>). and Zhou et&#xa0;al (<xref ref-type="bibr" rid="B29">Zhou et&#xa0;al., 2021</xref>). proposed an improved masked RCNN algorithm to identify cherry tomatoes by modifying the input layer of the network. Therefore, they performed bimodal data fusion of RGB and depth images, yielding good cherry tomato recognition results in the case where the fruit is adhered to the stem. However, this technique does not investigate the fruit diameter and is not able to meet the recognition requirements of the current study. Moreover, Yang et&#xa0;al (<xref ref-type="bibr" rid="B21">Yang et&#xa0;al., 2022</xref>). proposed a fast recognition algorithm for multi-apple targets in dense scenes, applying an improved Center Net model with the absence of an anchor frame to achieve accurate detection of apples in dense scenes; however, this method did not identify and detect fruits in the case of branch and leaf occlusion. Furthermore, Song et&#xa0;al (<xref ref-type="bibr" rid="B15">Song et&#xa0;al., 2022</xref>). introduced a fast and accurate localization algorithm for oil tea fruits in a natural and complex scene. Although this fruit is small, densely distributed, and colorful, the accuracy of the YOLOv5 Convolutional Neural Network (CNN) algorithm was high, but the algorithm&#x2019;s performance was limited in uneven light environments.</p>
<p>Regarding the Mask RCNN neural network model (<xref ref-type="bibr" rid="B2">Blok et&#xa0;al., 2021</xref>), it is mostly applied in detecting strawberries, apples, and pears. For instance, Chen et&#xa0;al (<xref ref-type="bibr" rid="B3">Chen et&#xa0;al., 2022</xref>). proposed a novel citrus fruit ripeness method combining visual saliency and CNNs. Initially, this method recognizes citrus fruits in an image using the YOLOv5 algorithm. Then, the visual saliency detection algorithm is improved, and a saliency map of the fruit is generated. Consequently, a four-channel ResNet34 network is utilized to combine the RGB image information with the saliency map in order to determine the fruit ripeness level. As a result, the detection accuracy is better; however, this method does not recognize and detect the fruits in the case of branch and leaf occlusion, presenting an important limitation. In addition, Pan et&#xa0;al (<xref ref-type="bibr" rid="B14">Pan and Ahamed, 2022</xref>). applied the Mask RCNN model combined with a three-dimensional (3D) stereoscopic camera to detect balsam pears in complex orchard environments. Consequently, the accuracy of the target detection in the validation set and the test set was relatively high; yet, in the case of picking scenarios, considering obstructed, overlapped, and unevenly illuminated fruits, the detection accuracy decreased sharply, making it difficult to meet the requirements of automated operation.</p>
<p>Therefore, to address the issue regarding fruits being obscured, overlapped, and unevenly illuminated, as well as the adherence to complex environments, this study used persimmons as the detection target. Then, we proposed a target detection method based on the improved Mask RCNN model. This technique consists of integrating cropping, morphological processing, and concave-point segmentation modules in the fully connected layer of the Mask RCNN network structure to reach the target detection in scenarios of fruits being, or not, obstructed by branches and leaves, fruit overlapping, and other scenarios of persimmon accurate detection. As a result, the proposed method targets provide theoretical and technical support for the development of intelligent picking robot fruit target detection technology.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Design of the recognition algorithm</title>
<sec id="s2_1">
<label>2.1</label>
<title>Color space model</title>
<p>The color feature is one of the most intuitive global features describing the target object (<xref ref-type="bibr" rid="B11">Liu et&#xa0;al., 2023b</xref>). It is frequently applied in orchard robots, where the color space model is a representation of different color component scales through coordinate axis parameters. Since the color change of the full life cycle of persimmon ranges from green to red, whereas the color change of leaves is from green to yellow, this study adopted the (L*a*b*color) space model based on comparisons and analysis of a variety of color space models. This representation is the most expressive recognition of the persimmon&#x2019;s full life cycle color features. The L*a*b* color space model provides a comprehensive, reliable, 3D color space with significant scalability to meet a variety of different color needs without being affected by any external environment and surpassing the traditional visual recognition methods. Therefore, this model achieves more accurate color recognition (<xref ref-type="bibr" rid="B7">Huang et&#xa0;al., 2023</xref>), as shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. Furthermore, the model stereogram was displayed in <xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f4">
<bold>4</bold>
</xref>, where the L* channel represents different degrees of light intensity, the value range is (0, 100), and the boundary of color ranges between black and white; the a* color changes between red and green; the b* color changes between yellow and blue, where the a* and b* value ranges are between &#x2212;128 and 127.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>L*a*b* color space model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g001.tif">
<alt-text content-type="machine-generated">A 3D sphere diagram representing the CIELAB color space. The vertical axis shows lightness, with white at the top and black at the bottom. The horizontal axes display color dimensions: green (-a*), red (+a*), blue (-b*), and yellow (+b*). Dashed lines indicate different color planes within the sphere.</alt-text>
</graphic>
</fig>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Median filter processing diagram. <bold>(a)</bold> Original, <bold>(b)</bold> median filter plot.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g002.tif">
<alt-text content-type="machine-generated">Two images depict fruit on a tree. The left image is in color, showing yellow fruit among green leaves. The right image is a black-and-white version with a median filter effect applied, creating a smoother appearance.</alt-text>
</graphic>
</fig>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Mask RCNN network process.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g003.tif">
<alt-text content-type="machine-generated">Diagram explaining a machine learning flow for image processing, featuring convolutional layers for feature extraction from an input image. The process includes Region Proposal Network (RPN) for objectness classification and bounding box regression. Feature maps with projected region proposals undergo RoI pooling, leading to multi-class classification and bounding box regression for each region of interest.</alt-text>
</graphic>
</fig>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Visualization effect of target detection and mask segmentation. <bold>(a)</bold> Original, <bold>(b)</bold> Visualization of mask segmentation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g004.tif">
<alt-text content-type="machine-generated">Left panel shows persimmons on a branch with leaves, in their natural environment. Right panel displays the same persimmons isolated on a black background, highlighting segmentation for analysis.</alt-text>
</graphic>
</fig>
<p>Regarding the study of fruit recognition and classification, the purpose of the color space transformation is to determine the appropriate color component and construct an effective color operator (<xref ref-type="bibr" rid="B24">Yu et&#xa0;al., 2023</xref>). The three-component gray scale example map of persimmon samples in L*a*b*space was shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The spatial transformation relationship between L*a*b*color and RGB color is nonlinear, whereas the XYZ color space is deployed as a bridge to achieve the spatial transformation between both. The conversion relationship between the three spaces is defined as follows:</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Three-primary color component plot of persimmon samples in L*a*b* space.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">L*a*b*mold</th>
<th valign="middle" align="left">L*color component</th>
<th valign="top" align="left">a*color component</th>
<th valign="top" align="left">b*color component</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-i001.tif">
<alt-text content-type="machine-generated">A round object with a smooth surface appears against a reddish-pink background. The object has a bluish hue with faint darker spots scattered across it.</alt-text>
</inline-graphic>
</td>
<td valign="top" align="center">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-i002.tif">
<alt-text content-type="machine-generated">Round, smooth fruit with minor blemishes in black and white.</alt-text>
</inline-graphic>
</td>
<td valign="top" align="center">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-i003.tif">
<alt-text content-type="machine-generated">Grayscale image of a round celestial body with varying shades and textures. Dark patches intermingle with lighter areas, creating a mottled appearance. The image lacks distinctive features that might identify the object.</alt-text>
</inline-graphic>
</td>
<td valign="top" align="center">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-i004.tif">
<alt-text content-type="machine-generated">A black and white image of a potato with several dark spots and blemishes on its surface, suggesting imperfections or signs of decay.</alt-text>
</inline-graphic>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Changing the RGB color space to an XYZ color space yields in the following (<xref ref-type="disp-formula" rid="eq1">
<bold>Equation 1</bold>
</xref>):</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mi>X</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>Y</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>Z</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>0.17697</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0.49</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0.31</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0.20</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0.17697</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0.8124</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0.01063</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0.00</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0.01</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0.99</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mi>R</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>G</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>B</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The XYZ color space is further varied into the <italic>L*a*b*color</italic> space where it is expressed as follows (<xref ref-type="disp-formula" rid="eq2">
<bold>Equation 2</bold>
</xref>):</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable columnalign="left" equalrows="true" equalcolumns="true">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>116</mml:mn>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>16</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>500</mml:mn>
<mml:mo>[</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mn>0.982</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>200</mml:mn>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mn>1.192</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The formula shown in <xref ref-type="disp-formula" rid="eq3">
<bold>Equation 3</bold>
</xref>:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>{</mml:mo>
<mml:mtable columnalign="left" equalrows="true" equalcolumns="true">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>13</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&gt;</mml:mo>
<mml:mn>0.008856</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mn>7.787</mml:mn>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>0.138</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>0.008856</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Design of persimmon identification method</title>
<p>Shooting in different weather and angles may cause shadows and noise in the image, therefore impacting the image recognition of the fruit. In this study, the median filter processing was applied to complete the denoising process, and the median filter processed image was depicted in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2b</bold>
</xref>.</p>
<p>After denoising the image through morphology, concave point segmentation, and filtering, the persimmon fruits are recognized by template matching algorithm. This algorithm is processed by spatially aligning the sensor to the acquired image under different conditions. Therefore, the template is a known small image and template matching consists of searching for a target image among the known small images, where both target and template have the same size and orientation, and the image is processed using a specific algorithm to find its target and determine its coordinate position (<xref ref-type="bibr" rid="B8">Jiang et&#xa0;al., 2023</xref>).</p>
<p>Moreover, the algorithm is designed in the following way: the search template <italic>T</italic> is superimposed on the searched map <italic>S</italic> (W*H pixels) to translate, and the area where the template covers the searched map is called the sub-map <italic>S<sub>ij</sub>
</italic>, where <italic>i</italic> and <italic>j</italic> represent the coordinates of the top left corner of the sub-map on the searched map S. The search range was shown in <xref ref-type="disp-formula" rid="eq4">Equations 4</xref>, <xref ref-type="disp-formula" rid="eq5">5</xref> as follows:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>-</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The process of template matching is achieved by comparing the similarity of <italic>T</italic> and <italic>S<sub>ij</sub>
</italic>. To measure the degree of matching between both entities, the following two measures can be applied, shown in <xref ref-type="disp-formula" rid="eq6">
<bold>Equations 6</bold>
</xref>, <xref ref-type="disp-formula" rid="eq7">
<bold>7</bold>
</xref>:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Improvement of instance segmentation algorithm for mask RCNN</title>
<sec id="s3_1">
<label>3.1</label>
<title>Mask RCNN algorithm improvement method</title>
<p>Mask RCNN modifies the network structure of Faster RCNN by adjusting two aspects of Region of Interest (RoI) Pooling and adding target mask branches to realize target pixel-level segmentation (<xref ref-type="bibr" rid="B12">Lv et&#xa0;al., 2023</xref>). Initially, the quantization operation of RoI Pooling is replaced with RoI Align of the linear interpolation algorithm to achieve accurate point-to-point alignment in the feature mapping process while maintaining the precise spatial location. Consequently, the proposed architecture integrates the predicted target mask branch in parallel to the original basis. Through this approach, binary mask images of all persimmon targets in the image are generated using Fully Convolutional Networks (FCNs) with inverse convolution (<xref ref-type="bibr" rid="B26">Zhang et&#xa0;al., 2023</xref>). Finally, segmentation masks and segmented persimmon images of varying ripeness levels are predicted and obtained in a pixel-to-pixel manner. This simultaneous execution of target detection and segmentation occurs across parallel branches, with separate branches handling classification and target detection frame regression concurrently.</p>
<p>The Mask RCNN algorithm empowers the final model to perform not only target detection and classification but also instance segmentation. This capability arises from its ability to accurately preserve the spatial location of pixels and achieve pixel-by-pixel mask prediction. The total loss function in Mask RCNN comprises three main components: the classification loss of candidate frames, location regression loss, and target mask loss, defined as shown in <xref ref-type="disp-formula" rid="eq8">Equations 8</xref>&#x2013;<xref ref-type="disp-formula" rid="eq12">
<bold>12</bold>
</xref>
</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the classification loss function, <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the location regression loss function, and <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the target mask loss function.</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the normalization parameters, <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicates the probability that the RoI with serial number is predicted to be a positive sample. Moreover, if <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the suggested regions are positive samples, whereas, when <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the suggested regions are negative samples.</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>{</mml:mo>
<mml:mtable columnalign="left" equalrows="true" equalcolumns="true">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>|</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>|</mml:mo>
<mml:mo>&lt;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>|</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the normalized parameters, <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the offset of the prediction box, <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msubsup>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
</mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> highlights the parameters for the actual offset, and <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>&#x2014;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> indicates the loss value.</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msup>
<mml:mi>m</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">[</mml:mo>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mi>log</mml:mi>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the target true label value, <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> indicates the predicted values, and show the number of target categories for the instance segmentation task.</p>
<p>The network flow of Mask RCNN was illustrated in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, and its algorithm process steps are the following: first, the image pre-processed picture is inputted to the Resnet50 feature extraction network pre-trained by mitigating learning to obtain the feature mapping map. Then, the predetermined RoIs are set in the feature map to generate multiple ROIs. Consequently, the first before and after scene classification are carried out in the Region Proposal Network (RPN) to detect if there is a target background in the prediction frames, and perform some correction on the prediction frames, such as filtering out part of the useless prediction frames. Finally, the resulting candidate frames are subjected to RoI Align operation, suggesting an output layer of prediction class labels, bounding boxes, and target masks for the region.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Example segmentation algorithm improvement method</title>
<p>Instance segmentation serves as a combination of two methods, semantic segmentation and target detection (<xref ref-type="bibr" rid="B23">Yi and Wang, 2023</xref>; <xref ref-type="bibr" rid="B19">Wang et&#xa0;al., 2024</xref>). The former consists of dividing the image or video, according to category similarities and differences, into multiple blocks, which are then transformed into machine language, realizing the classification of the image at the pixel level. As for the latter, it consists of distinguishing all target objects of interest in an image and determining their categories and locations, as various types of objects have different appearances, shapes, and postures, as well as distinct interference factors such as illumination and occlusion during imaging. This results in target detection difficulties and yields several problems in classification, localization, detection, and segmentation.</p>
<p>The requirement of instance segmentation consists of performing target detection based on semantic segmentation requirements, that is, to distinguish each instance target based on predicting the target contour for each pixel category (<xref ref-type="bibr" rid="B22">Ye et&#xa0;al., 2023</xref>). Therefore, this study serves to recognize fruit diameter detection of persimmons in a complex orchard environment, requiring high recognition and segmentation accuracy. Therefore, this study utilizes the expandability of the Mask RCNN algorithm to accurately achieve persimmon fruit diameter detection in complex orchard environments by instance segmentation based on target detection. Finally, the Mask RCNN algorithm&#x2019;s target detection of persimmons and mask segmentation visualization results were illustrated in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>.</p>
<p>The visualization effect of mask segmentation can be found in the picture. While the mask RCNN instance segmentation algorithm for persimmon fruit generally outlines the location accurately, it exhibits less precision at the edges of the fruit segmentation. Additionally, some occlusion objects within the fruit edge segmentation obscure accurate reflection and location of the persimmon fruit&#x2019;s edge information. Further optimization of the segmentation process is necessary to enhance the accuracy of edge detection and localization.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Design of fruit instance segmentation network architecture based on improved Mask RCNN</title>
<p>The mask in the Mask RCNN algorithm relies on the persimmon recognition detection module, where the mask branch is acquired through a cropping process based on the prediction frame and specific thresholds. However, the masks obtained for fruit instances using this method exhibit limited differentiation ability, particularly at the boundaries of objects of the same species. This becomes more pronounced when fruits occlude each other, leading to competition among edge pixels.</p>
<p>Furthermore, the fruit target detection frame generated by the RoI feature in the fully connected layer may include images of objects other than fruits. Thus, another cropping process is applied to extract individual fruit target images. However, when fruits are partially occluded by branches or leaves, or when they overlap with each other, thresholding the pixels at the boundary of the fruits during cropping can result in issues such as sticking to occluded objects or mixing with background features, thereby affecting the final segmentation results.</p>
<p>To mitigate the effects of occlusion, overlap, and adhesion, morphological processing and concave point segmentation operations are employed to obtain the accurate contour of the target fruit. The improved Mask RCNN instance segmentation network structure was illustrated in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Improved Mask RCNN instance segmentation network structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g005.tif">
<alt-text content-type="machine-generated">Flowchart depicting a neural network architecture for object detection. Starting with RoI Align, the image passes through FC layers, segmented into three branches: &#x201c;bbox reg&#x201d; to &#x201c;Coordinates&#x201d; and &#x201c;softmax&#x201d; to &#x201c;Category.&#x201d; The process involves improved modules: cropping, morphological treatment, and pit segmentation, leading to fully convolutional networks and generating a mask.</alt-text>
</graphic>
</fig>
<p>Morphological processing is a fundamental technique used for manipulating the shape features of an image to achieve specific objectives. It primarily involves operations such as expansion, erosion, and adhesion analysis. In more detail, the expansion operation consists of reading each pixel in the image one by one using a defined rectangular template and modify the value of the pixel to the maximum value to connect the salient points on the periphery of the image and extend them outwards. As for the expansion operation, it involves the process of finding the local maximum value by assigning the maximum value pixel point as a reference to the surrounding pixels to achieve the purpose of expanding the highlighted features in the image. The expanded image has a larger target area compared to the original image. Finally, the erosion operation has an opposite effect to expansion operation, that is, the process of obtaining the local minimum value. The assignment of the minimum value to the surrounding pixel points yielding in making the highlighted part of the image shrink; as a result, the target area shrinkage is reflected in the visual effect, and the location and area size of the fruit adhesion are generated by extracting the adhesion component. This method consists of using a rectangle as a template, reading all the pixels covered by the template one by one, modifying the value of pixel X to the smallest value among all the pixels, and corroding the highlights in the periphery of the image.</p>
<p>After performing the fruit example cropping, there are still some adhesions between the fruit surrounding and branches, leaves, and other fruits. Consequently, the cropped image is converted into a binarized image to determine the location and area of the adhesion image (<xref ref-type="bibr" rid="B28">Zhou et&#xa0;al., 2022</xref>); then, small ranges of adhesion areas are directly deleted.</p>
<p>The target and background separation was basically solved by morphological processing; however, there were many problems of mutual occlusion and adhesion of persimmon fruits in natural environment, as shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6a</bold>
</xref>. The image after the morphological processing was illustrated in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6b</bold>
</xref>. Therefore, in the segmentation process of fruit instances, the problem of adhesion caused by mutual occlusion between fruits at their edge still resides. To tackle it, a solution consists of using concave point-based segmentation to find the concave points at the location of adhesion of fruits. Once found, these concave points will be connected so as to complete a more detailed segmentation of the target fruits.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Processed pictures. <bold>(a)</bold> The target cropped picture, <bold>(b)</bold> Morphological treatment.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g006.tif">
<alt-text content-type="machine-generated">(a) Close-up of a fruit with a yellowish-orange color, overlaid with white rectangular lines and text indicating &#x201c;Eight: 96%&#x201d; and &#x201c;Eight: 94%&#x201d;. (b) Black image with white outline of the same fruit shape, illustrating morphological treatment.</alt-text>
</graphic>
</fig>
<p>As the first step of concave point segmentation consists of finding the exact location of the concave point, this study adopts the vector pinch method for point extraction. This method considers each pixel point on the edge of the target object as a corner point, connected to front loci and back loci. These loci are also on the edge of the target object and at the same distance from the corner point; moreover, they are set pre- and post-determination of the corner point, as shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. The angle between the two straight lines formed by the front locus and the corner point and the back locus and the corner point, denoted as, and expressed in <xref ref-type="disp-formula" rid="eq13">
<bold>Equations 13</bold>
</xref>&#x2013;<xref ref-type="disp-formula" rid="eq16">16</xref> for concave point extraction.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Diagram of the concave point segmentation process. <bold>(a)</bold> Vectorial pinch angle, <bold>(b)</bold> Concave point extraction, <bold>(c)</bold> Dividing line.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g007.tif">
<alt-text content-type="machine-generated">Three-panel diagram illustrating vector analysis. (a) A vectorial pinch angle is shown with vectors A, B, and C forming an angle. (b) Concave point extraction is depicted with two concave points marked by crosses on a curved line. (c) A dividing line is represented by a dashed line bisecting the curved shape.</alt-text>
</graphic>
</fig>
<p>An excessively high threshold may misidentify minor contour fluctuations as concave points, while an excessively low threshold may fail to detect shallow adhesion regions. In local coordinate systems, concave points exhibit negative curvature, typically corresponding to negative angular ranges (&#x2212;50&#xb0; to 0&#xb0;), whereas convex points correspond to positive angles (0&#xb0; to 50&#xb0;). Therefore, this paper selecting [&#x2212;50&#xb0;, 50&#xb0;] effectively covers typical adhesion-induced concavities, and it was shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7a</bold>
</xref>.</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>|</mml:mo>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>B</mml:mi>
<mml:mo>|</mml:mo>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>B</mml:mi>
<mml:mo>|</mml:mo>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>cos</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>|</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:msup>
<mml:mo>|</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mo>|</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>B</mml:mi>
<mml:msup>
<mml:mo>|</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mo>|</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>B</mml:mi>
<mml:msup>
<mml:mo>|</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>|</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>|</mml:mo>
<mml:mo>|</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>B</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>180</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The concave points are determined by the concave point segmentation formula as shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7b</bold>
</xref>. Consequently, the concave points are matched and connected to form the segmentation line. During the segmentation process, the pairs of concave points that conform to the directional characteristics of the circular growth of the fruit and have the smallest Euclidean distance between two pairs of concave points are matched and connected.</p>
<p>This study employs the Douglas-Peucker algorithm for polygonal contour approximation while preserving principal geometric features, implementing a comprehensive processing pipeline that includes (1) noise reduction through local curvature filtering, where the curvature (angle change per arc length) at each point is calculated to eliminate outliers with abrupt variations; (2) false segmentation prevention via dual geometric constraints on concave points, enforcing inter-point spacing (Rmin &#x2264; d &lt; 2Rmax) and recession depth (h &gt; 0.2R) to exclude shallow depressions caused by leaf occlusion; and (3) topological consistency validation requiring segmented sub-contours to satisfy both area thresholds (A &#x2208; [0.7Aavg, 1.34Aavg], where Aavg represents the average single-fruit area) and circularity criteria (4&#x3c0;A/P&#xb2; &gt; 0.7, with P denoting perimeter, effectively filtering non-fruit fragments). Therefore, the segmentation line is finally determined, as displayed in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7c</bold>
</xref>.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Fruit diameter detection method design</title>
<sec id="s4_1">
<label>4.1</label>
<title>Fruit edge information extraction</title>
<p>After getting a finer image segmentation after performing morphological processing, the fruit edge coordinate information is extracted. Moreover, the mask pixel point coordinates, obtained after segmentation, are first saved in the extraction process, followed by the edge coordinates. In the Mask RCNN algorithm, the pixel coordinates in the output image of the mask are &#x201c;True&#x201d; when the mask image yields the target object and &#x201c;False&#x201d; when the mask image is the background image. Considering the upper left corner of the image as the origin to establish the (<italic>X</italic>, <italic>Y</italic>) axis coordinate system, take the <italic>X</italic> axis as the base. By moving the straight line <italic>y</italic> = <italic>i</italic> parallel to the <italic>X</italic> axis, this line is traversed from left to right. After achieving the retrieval, the line is panned from top to bottom, as highlighted in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7a</bold>
</xref>. According to the above method, all pixel coordinates in the retrieved image are gradually traversed, and, finally, all pixel coordinates of the target area in the mask are set as &#x201c;True.&#x201d;</p>
<p>Following the implementation of this method, only the fruit mask edge coordinate information is preserved in order to detect the size of the fruit. While saving the coordinates of all pixels, the information in the image is also stored from left to right by traversing the line parallel to the <italic>X</italic> axis. Consequently, for each line parallel to the <italic>X</italic> axis, the first and the last pixel coordinates &#x201c;True&#x201d; correspond to both edges of the fruit, and the cycle of this method acquires all pixel coordinates of the fruit edges. The visualization results were shown in <xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8a, b</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Edge information processing diagram. <bold>(a)</bold> How mask coordinates are retrieved, <bold>(b)</bold> Fruit edge extraction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g008.tif">
<alt-text content-type="machine-generated">(a) A photograph showing a round fruit with overlaid axes and labeled mask coordinates, highlighting how the region is identified with 80% accuracy. (b) A graphic depicting the fruit's outline on a black background, illustrating edge extraction.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Fruit diameter detection method</title>
<p>Before measuring the fruit diameter, the coordinates of the center of the persimmon fruit image are generated, where the maximum connection length through the center of the shape represents the width of the persimmon fruit and the shortest connection length through the center of the shape indicates the thickness of the fruit. Based on this definition, the calculation begins by traversing the pixel points along the edge of the fruit. Then, the distance between each pair of consecutive coordinates is computed using the following expression, shown in <xref ref-type="disp-formula" rid="eq17">
<bold>Equation 17</bold>
</xref>:</p>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The process continues by iteratively calculating the distance equation between each pair of consecutive coordinates. During the iteration, the algorithm prioritizes the coordinates with the longest distance between them. At the end of the cycle, the fruit&#x2019;s edge is determined by the line connecting the two points of the longest connection with the shortest connection serving as the fruit diameter width and thickness. It is important to mention that the four Pixel coordinate points <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> were saved for further analysis.</p>
<p>The persimmon fruit diameter size measurements are obtained by applying this calculation. In this study, the camera was calibrated using the Zhang Zhengyou calibration method, converting the acquired pixel point coordinates into real 3D spatial points. The conversion relation for obtaining the spatial point <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> with pixel coordinate point <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> was represented in <xref ref-type="disp-formula" rid="eq18">Equation 18</xref>.</p>
<disp-formula id="eq18">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>[</mml:mo>
<mml:mtable columnalign="left" equalrows="true" equalcolumns="true">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mi>x</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mi>y</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mi>&#x3b1;</mml:mi>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mi>&#x3b2;</mml:mi>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bd;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mi>R</mml:mi>
</mml:mtd>
<mml:mtd>
<mml:mi>T</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mn>0</mml:mn>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>[</mml:mo>
<mml:mtable columnalign="left" equalrows="true" equalcolumns="true">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>represents the vertical axis of the camera&#x2019;s spatial coordinate system, demote the external parameters of the camera, and <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3c5;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicate the internal parameters of the camera.</p>
<p>According to the Zhang Zhengyou calibration method, to achieve the camera left and right sides of the respective single target calibration, the conversion relationship is represented as follows (<xref ref-type="disp-formula" rid="eq19">
<bold>Equation 19</bold>
</xref>):</p>
<disp-formula id="eq19">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mtable columnalign="left" equalrows="true" equalcolumns="true">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>R</mml:mi>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>[</mml:mo>
<mml:mtable columnalign="left" equalrows="true" equalcolumns="true">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>R</mml:mi>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> represents the points on the imaging plane of the left camera, <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denote the point on the imaging plane of the right camera, <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicate the external parameters of the left camera, <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> highlight the right camera external parameters.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Test results and analysis</title>
<sec id="s5_1">
<label>5.1</label>
<title>Purpose and methodology of the test</title>
<p>To test the effectiveness of the improved instance segmentation algorithm, this study initially analyzed and compared the effectiveness of the modules added to the Mask RCNN instance segmentation using the ablation test, as well as quantitatively analyzed the role of each module. Consequently, the comparison test of the fruit diameter measurement was performed to verify the efficiency of the improved segmentation and realize the automatic calculation and measurement of the program.</p>
<p>This study builds upon previous research (<xref ref-type="bibr" rid="B10">Liu et&#xa0;al., 2023a</xref>), utilizing a training set of 9,300 persimmon images at different growth stages. Additionally, 900 new persimmon images were collected from the National Persimmon Germplasm Repository in Hefei, Anhui Province, covering various growth phases. Through data augmentation techniques&#x2014;including random cropping, random flipping, random contrast enhancement, and color variation&#x2014;the newly acquired 900-image dataset was expanded to 3,000 images to serve as the test set.</p>
<p>Ablation test method: by means of module superposition, that is, the control variable method, the improved modules are combined in different arrangements, and the test was carried out sequentially. The test platform is set up as shown in Section 4.3, using MIoU, mean Average Precision (mAP) value, and Frames Per Second (FPS) as evaluation indexes, and applied to analyze and compare the instance segmentation impact of the improved algorithm.</p>
<p>The samples of the 200 were randomly divided into five groups, where the images of Group 1 were detected using the basic Mask RCNN algorithm, then Groups 2, 3, and 4 using three algorithms (e.g., the cropping processing module, the morphological processing module, and the concave point segmentation module, respectively), and Group 5 using the improved Mask RCNN algorithm. Each group of tests was repeated three times, and the Average Precision (AP) and mAP were recorded. Moreover, the average value of the three times was recorded as the valid value. The results of these three algorithms were compared to make sure that the proposed Mask RCNN algorithm can accurately recognize the target.</p>
<p>Fruit diameter measurement test method: 200 images were randomly selected in a natural environment without being covered by branches and leaves or other objects of the fruit as a test sample. Moreover, vernier calipers were used to manually detect the same this, to obtain the fruit diameter measurement width <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>&#xa0;K</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>z</mml:mtext>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and thickness <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>&#xa0;K</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>z</mml:mtext>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. According to the size, fruits were divided into 10 groups. In addition, a test through the split detection was performed to obtain the fruit diameter size width <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:msub>
<mml:mtext>K</mml:mtext>
<mml:mrow>
<mml:mtext>c</mml:mtext>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and thickness <inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:msub>
<mml:mtext>K</mml:mtext>
<mml:mrow>
<mml:mtext>c</mml:mtext>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and compared with the manual measurement value. Each group consisted of an average of 20 fruits recorded. Through analysis, the size of the error value was computed to test the accuracy of the improved algorithm in this study.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Data acquisition and test bench construction</title>
<p>To meet the diversity of persimmon literacy growth in complex environments, the collected images fully considered the variability of the sample data; that is, 3,300 persimmons in different growth periods were collected in variable environments such as using different light conditions, different angles, different numbers of fruits, branch leaf shade, and overlapping of multiple clusters of fruits, as shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Example of a partial image sample of a persimmon fruit in nature. <bold>(a)</bold> Overlapping fruits shaded by foliage, <bold>(b)</bold> Shade, <bold>(c)</bold> Fruits are independent of each other, <bold>(d)</bold> Exposure, <bold>(e)</bold> Cloudy day and smooth light, <bold>(f)</bold> Backlighting on a sunny day.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g009.tif">
<alt-text content-type="machine-generated">Six images showing different lighting conditions for fruit on trees. a. Fruits overlap, shaded by foliage. b. Fruits in shade. c. Fruits spaced apart. d. Fruits in bright exposure. e. Fruits under cloudy, smooth light. f. Fruits with backlighting on a sunny day.</alt-text>
</graphic>
</fig>
<p>The specific operating environment parameters of this study were displayed in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. Moreover, two sets of self-constructed persimmon datasets were randomly assigned to the training and test sets using a 9:1 ratio.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Runtime environment parameters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Hardware</th>
<th valign="top" align="left">Configuration</th>
<th valign="top" align="left">Environment</th>
<th valign="top" align="left">Version</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">CPU</td>
<td valign="top" align="left">Intel Core i7-12700</td>
<td valign="top" align="left">Python</td>
<td valign="top" align="left">3.7.13</td>
</tr>
<tr>
<td valign="top" align="left">GPU</td>
<td valign="top" align="left">RTX 3060</td>
<td valign="top" align="left">PyTorch</td>
<td valign="top" align="left">1.7.1+cu110</td>
</tr>
<tr>
<td valign="top" align="left">RAM</td>
<td valign="top" align="left">64 G</td>
<td valign="top" align="left">CUDA</td>
<td valign="top" align="left">11.0</td>
</tr>
<tr>
<td valign="top" align="left">Hard-disk</td>
<td valign="top" align="left">520G</td>
<td valign="top" align="left">CUDNN</td>
<td valign="top" align="left">8.0.5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To avoid problems such as overfitting or underfitting caused by improper tuning of hyper-parameters for model training, the tuning of hyper-parameters was achieved by the network searching method to obtain optimal numerical points by giving a larger range of data and smaller searching step size. In this study, the number of iteration rounds (epoch) was set to 300, the learning rate (Ir) was set to 0.01, the optimizer adopts Stochastic Gradient Descent (SGD), and the momentum factor (momentum) parameter was set to 0.9. During the training period, the tensor-board was employed to record the loss and learning rate of the training set generated by each iteration. Data, including the training set loss and learning rate changes generated, and weights were kept and saved.</p>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Evaluation indicators</title>
<p>
<bold>(1) <italic>Precision (P</italic>
</bold>): for the model prediction results, it represents the number of positive cases in the prediction process, that is, the ratio of the number of correct objects detected to the total number of correct objects in the sample. Moreover, the response category prediction correctness is employed as a measure of the accuracy for the model detection, shown in <xref ref-type="disp-formula" rid="eq20">
<bold>Equation 20</bold>
</xref>.</p>
<disp-formula id="eq20">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:mtext>P</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the genuine example and <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the False Positive example.</p>
<p>
<bold>(2)<italic>Recall (R)</italic>
</bold>: for all the samples in the dataset, it indicates the number many positive cases that are correctly predicted, that is, the ratio of the number of correct objects detected to the number of objects in the sample. It also measures the number of positive samples obtained by the model through the prediction process, shown in <xref ref-type="disp-formula" rid="eq21">
<bold>Equation 21</bold>
</xref>.</p>
<disp-formula id="eq21">
<label>(21)</label>
<mml:math display="block" id="M21">
<mml:mrow>
<mml:mtext>R</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> represents the False Negative example.</p>
<p>
<bold>(3)<italic>P-R curve</italic>
</bold>: a graph constituted by connecting the corresponding points of the horizontal and vertical coordinates indicated through the recall and the precision rates. Furthermore, a corresponding P-R curve is built for each category in the prediction process.</p>
<p>
<bold>(4)<italic>Average-Precision (AP</italic>
</bold>): it represents the area of the <italic>PR</italic> curve plotted by <italic>P</italic> and <italic>R</italic>. The area under the curve denotes the average of all the accuracies across recall values. Moreover, it measures the accuracy of the model in given categories, shown in <xref ref-type="disp-formula" rid="eq22">
<bold>Equation 22</bold>
</xref>.</p>
<disp-formula id="eq22">
<label>(22)</label>
<mml:math display="block" id="M22">
<mml:mrow>
<mml:mtext>AP</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where i in number of thresholds and j represents the category.</p>
<p>
<bold>(5)<italic>Mean Average Precision (mAP)</italic>
</bold>: it represents the average value of AP in each category, measuring the accuracy value of the trained model in all categories. Moreover, it denotes the most important evaluation index in target detection algorithms, shown in <xref ref-type="disp-formula" rid="eq23">
<bold>Equation 23</bold>
</xref>.</p>
<disp-formula id="eq23">
<label>(23)</label>
<mml:math display="block" id="M23">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>n</italic> represents the total number of categories.</p>
<p>
<bold>(6)<italic>Mean Intersection-over-Union (MIoU)</italic>
</bold>: it represents the most representative evaluation index for segmentation networks. It describes the overlapping degree between the candidate frames and the manually labeled frames generated by all categories during the model training process. Moreover, it calculates the average value of the ratio of the resulting intersection and concatenation to quantify the fitting degree between both frames and then to judge the quality of the model detection, shown in <xref ref-type="disp-formula" rid="eq24">
<bold>Equation 24</bold>
</xref>.</p>
<disp-formula id="eq24">
<label>(24)</label>
<mml:math display="block" id="M24">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where 1 represents the true value.</p>
<p>
<bold>(7)<italic>FPS</italic>
</bold>: it represents the number of images that can be processed per second by the network model, being an evaluation index to reflect the real-time performance of the model.</p>
<p>
<bold>(8)<italic>Relative error</italic>
</bold>, shown in <xref ref-type="disp-formula" rid="eq25">
<bold>Equations 25</bold>
</xref>, <xref ref-type="disp-formula" rid="eq26">
<bold>26</bold>
</xref>:</p>
<disp-formula id="eq25">
<label>(25)</label>
<mml:math display="block" id="M25">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq26">
<label>(26)</label>
<mml:math display="block" id="M26">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im33">
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> represents the width of the fruit diameter, <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> denotes the thickness of the fruit diameter, <inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> indicates the actual fruit diameter width, and, finally, <inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> yields the Actual fruit thickness.</p>
</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Ablation test results and analysis</title>
<p>Referring to <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, the MIoU values of the three network algorithms after the integration of the cropping processing module, morphological processing module, and concave point segmentation module to the base network Mask RCNN were set to 82.9%, 86.4%, and 88.8%, respectively; as for the mAP values, they were set to be 88.49%, 90.67%, and 91.95%, respectively. The three indexes of adding the cropping processing module compared to the basic network are improved by 11.73%, 2.66%, and 11.76%, respectively, proving that the addition of the cropping processing module can significantly enhance the performance of the network in terms of the mean intersection, parallel ratio, and processing speed. However, the enhancement of the average accuracy remains relatively small.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Ablation test data chart.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g010.tif">
<alt-text content-type="machine-generated">Bar chart comparing different segmentation methods with metrics: mIoU%, mAP%, and FPS. The Improved Instance Segmentation Network scores highest in mIoU% (92.7) and mAP% (94.25). Cropping Processing has the highest FPS at 76.</alt-text>
</graphic>
</fig>
<p>In addition, the three metrics of adding the morphological processing module compared to the base network are improved by 16.44%, 5.19%, and 2.94%, respectively, showing that adding the morphological processing module can significantly improve the performance of the network regarding the equalization and concatenation ratio. Moreover, the improvement of the average accuracy is relatively large, but the enhancement of the performance of the processing speed is relatively small. All three indexes of adding the concave point segmentation module compared to the base network are improved by 19.68%, 6.67%, and 5.88%, respectively, proving that the addition of the concave point segmentation module can significantly enhance the network&#x2019;s equalization and concurrency ratio performance. Meanwhile, the effect of the average accuracy improvement and processing speed performance is more significant. It is verified that the enhanced method proposed in this study for the base network Mask RCNN is efficient and effective.</p>
<p>Referring to <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, the improved algorithm in this study by simultaneously adding the three modules of cropping, morphological processing, and concave point segmentation has an MIoU value of 92.7%, an mAP value of 94.25%, and an FPS value of 74. Compared to the base network, the MIoU value has been increased by 24.93%, the mAP value has been improved by 9.34%, and the FPS value has been adjusted by 8.82%. This proves that the improvement method by adding the three modules to the base network Mask RCNN at the same time is effective. Compared to the base network, the performance of all aspects is improved, and the required accuracy of the fruit diameter measurement is achieved.</p>
</sec>
<sec id="s5_5">
<label>5.5</label>
<title>Fruit diameter measurement test results and analysis</title>
<p>(1) Comparative analysis of fruit diameter width</p>
<p>(2) Comparison of fruit diameter thickness analysis</p>
<p>The analysis of the test error results, illustrated in <xref ref-type="fig" rid="f11">
<bold>Figures&#xa0;11</bold>
</xref>, <xref ref-type="fig" rid="f12">
<bold>12</bold>
</xref>, shows that due to the autogenous growth characteristics of persimmon fruit, its edge is smooth and not easily deformed, yielding measurement difficulties. However, in the measurement of fruit width, the relative error varies between 0.17% and 1.3%, representing a small interval and proving that this study has a high accuracy in the measurement of fruit width. Concerning the thickness of persimmon fruit, it is difficult to measure it for fruit diameter due to the easy deformation of the fruit caused by the connection of the fruit stalk. However, while measuring it, the relative error varies between 1.01% and 1.98%, also representing a small interval, and proving, once again, this study had a high accuracy in the measurement of fruit thickness. While excluding the effect of deformed fruit and vernier caliper reading error, the relative error range in the overall measurement of fruit thickness and fruit width is within a reasonable acceptable range.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Comparison of data from some data and fruit width.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g011.tif">
<alt-text content-type="machine-generated">Bar graph comparing actual fruit diameter and test fruit diameter across ten groups, with a line graph showing decreasing relative error percentage. Orange bars represent actual diameter, green bars represent test diameter, and blue line indicates relative error.</alt-text>
</graphic>
</fig>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Comparison of some data and fruit thickness test data.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1636727-g012.tif">
<alt-text content-type="machine-generated">Bar and line chart showing fruit diameter and relative error across ten groups. Orange bars represent actual diameter, green bars show test diameter, and the blue line indicates relative error in percentage decreasing from group one to ten.</alt-text>
</graphic>
</fig>
<p>Based on the error curve, it is clear that, with the increase of persimmon fruit width or fruit thickness, the relative error is smaller. Due to the error index of this study that represents the relative error, the error decreases with the increase of fruit diameter. Therefore, it can be concluded that the measurement accuracy of fruit diameter is stable and unchanged, and the surface of the improved algorithm has a significant improvement in detection accuracy and precision performance.</p>
</sec>
<sec id="s5_6" sec-type="discussion">
<label>5.6</label>
<title>Discussion</title>
<p>Target detection has been the most challenging problem in the field of computer vision due to the different appearances, shapes, and poses of outdoor persimmons. Moreover, several environmental conditions, including the interference of light and occlusion, were considered as disturbances during imaging. Therefore, the addition of the cropping processing module, the morphological processing module, and the concave point segmentation module, respectively, yielded three new detection algorithms and resulted in improved Mask RCNN instance segmentation algorithm detection.</p>
<p>Referring to <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, the results of the ablation experiments indicate that, compared to the basic Mask RCNN instance segmentation algorithm, the improved Mask RCNN instance segmentation algorithm resulted in an enhancement of the MIoU value by 24.93%, the mAP value by 9.34%, and the FPS value by 8.82%. Therefore, the MIoU and mAP values are most significantly improved with the addition of clipping, morphological processing, and concave point segmentation modules, respectively; yet, the FPS metric improvement effect is missing and lower than the improvement rate when only the clipping module is added.</p>
<p>As the FPS index indicates the number of images processed per second by the network model, representing an evaluation index to reflect the real-time performance of the model is a must. In the fully connected layer of the basic Mask RCNN network structure, adding the cropping module will improve extracted image features and increase the model processing speed. However, when the clipping, morphological processing, and concave point segmentation modules are simultaneously integrated, the model becomes more replicated, therefore increasing the program running time, which in turn leads to a lower overall processing speed than when the clipping module is added alone.</p>
<p>Moreover, in this study, the vector clip angle method is applied for concave point extraction. This method consists of considering each pixel point on the edge of the target object as a corner point. Consequently, set two anterior and posterior loci, representing also the edge of the target object and at an equal distance from the corner point, before and after the determined corner point, and the anterior and posterior loci from two straight lines with the corner point, respectively, and the two straight lines form an angle, and then set the appropriate clip angle for the extraction of concave points applying <xref ref-type="disp-formula" rid="eq16">Equation 16</xref>. When the clipping module and the concave point segmentation module are added simultaneously, the clipping module will crop out some pixel points, reducing the pixel points at the edges of the target object and slowing down the concave point extraction speed. Therefore, the addition of the three modules at the same time will reduce the execution speed compared to the use of the clipping module alone.</p>
<p>According to <xref ref-type="fig" rid="f11">
<bold>Figures&#xa0;11</bold>
</xref>, <xref ref-type="fig" rid="f12">
<bold>12</bold>
</xref>, the relative error decreases with the increase of fruit diameter, and this decreasing trend of the relative error curve is approximately linear, showing that the improved Mask RCNN algorithm has stable measurement accuracy. Moreover, this study further segments the occluded objects, such as fruit stalks, by adding a morphological processing module and a concave point extraction processing method to improve the measurement accuracy. The enhanced Mask RCNN algorithm reduces the parameters for training, rendering the filter independent of the signal position; it also improves the characteristics of the detected signal and reinforces the generalization ability of the trained model. Finally, the proposed algorithm proves the correctness and effectiveness of the improved Mask RCNN instance segmentation method.</p>
</sec>
</sec>
<sec id="s6" sec-type="conclusions">
<label>6</label>
<title>Conclusion</title>
<p>In this study, focusing on the persimmon inspection recognition and fruit diameter detection in natural environments, challenges, such as fruit occlusion and adhesion, were addressed. Through enhancements made to the Mask RCNN instance segmentation algorithm, the aim was to achieve high-precision, non-destructive detection of fruit diameter.</p>
<p>This study leverages the versatility of the Mask RCNN algorithm to achieve accurate fruit recognition in intricate environments through instance segmentation following target detection. Additionally, an enhanced instance segmentation algorithm for Mask RCNN is proposed to achieve accurate fruit detection. Building upon the RCNN network structure, the RoI Pooling quantization operation is substituted by RoI Align employing a linear interpolation algorithm, and a parallel predictive target mask branch is introduced to enable accurate fruit recognition. Moreover, in this study, a fruit diameter detection method in complex environments was added to tackle the problem of edge pixel competition when fruits mask each other. Finally, integrating cropping, morphological processing, and concave point segmentation modules to the fully connected layer of the Mask RCNN network structure results in a precise contour detection of the target fruits and the identification of the fruit diameter measurement.</p>
<p>The results of the ablation test highlight that the improved algorithm has significantly improved the mAP, MIoU, and FPS values compared to the traditional Mask RCNN instance segmentation algorithm, verifying the validity of this study to significantly improve the measurement accuracy of the improved Mask RCNN instance segmentation algorithm by adding these different modules. Moreover, the results of the fruit diameter measurement test verify that the proposed method can effectively tackle the challenges of adhesion in the fruit measurement by adding the cropping, morphology processing, and concave point segmentation module to the Mask RCNN instance segmentation improved algorithm. As a result, high segmentation accuracy would solve the fruit measurement challenges in sticky problems and can accurately detect the size of the fruit diameter.</p>
<p>Finally, it is worth noting that this study provides theoretical support and technical reference for orchard inspection, yield estimation, fruit detection, and mechanized picking. This study enables prediction of batch harvesting schedules and yield estimation for persimmons based on market demand, thereby increasing growers&#x2019; income.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>YFa: Writing &#x2013; original draft, Formal Analysis, Software, Methodology, Data curation, Writing &#x2013; review &amp; editing. YFe: Writing &#x2013; original draft, Formal Analysis, Writing &#x2013; review &amp; editing, Data curation. YL: Formal Analysis, Resources, Writing &#x2013; original draft, Funding acquisition, Methodology, Data curation, Supervision, Investigation, Software, Writing &#x2013; review &amp; editing. YC: Validation, Investigation, Supervision, Writing &#x2013; review &amp; editing. HJ: Supervision, Conceptualization, Writing &#x2013; review &amp; editing, Funding acquisition, Visualization.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This research was funded by National Natural Science Foundation of China, grant number 32401687; and Scientific Research Projects of Universities in Anhui, grant number 2024AH050421.</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2025.1636727/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2025.1636727/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.zip" id="SM1" mimetype="application/zip"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname> <given-names>Y. H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>B. H.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>N. M.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J. Y.</given-names>
</name>
<name>
<surname>Diao</surname> <given-names>Z. H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Vision-based navigation and guidance for agricultural autonomous vehicles and robots: A review</article-title>. <source>Comput. Electron. Agric.</source> <volume>205</volume>, <elocation-id>107584</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.107584</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blok</surname> <given-names>P. M.</given-names>
</name>
<name>
<surname>van Evert</surname> <given-names>F. K.</given-names>
</name>
<name>
<surname>Tielen</surname> <given-names>A. P.</given-names>
</name>
<name>
<surname>van Henten</surname> <given-names>E. J.</given-names>
</name>
<name>
<surname>Kootstra</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The effect of data augmentation and network simplification on the image-based detection of broccoli heads with Mask RCNN</article-title>. <source>J. Field Robotics</source> <volume>38</volume>, <fpage>85</fpage>&#x2013;<lpage>104</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/rob.21975</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>J. T.</given-names>
</name>
<name>
<surname>Jiao</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Z. M.</given-names>
</name>
<name>
<surname>Huo</surname> <given-names>Z. W.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>W. X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Citrus fruits maturity detection in natural environments based on convolutional neural networks and visual saliency map</article-title>. <source>Precis. Agric.</source> <volume>23</volume>, <fpage>1515</fpage>&#x2013;<lpage>1531</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11119-022-09895-2</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Faria</surname>
<given-names>S. E. S.</given-names>
</name>
<name>
<surname>Azevedo</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Rabelo</surname> <given-names>N. G.</given-names>
</name>
<name>
<surname>Anast&#xe1;cio</surname> <given-names>V. Z.</given-names>
</name>
<name>
<surname>Maciel</surname> <given-names>V. M.</given-names>
</name>
<name>
<surname>Matos</surname> <given-names>D. V.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>Advanced phenotyping in tomato fruit classification through artificial intelligence</article-title>. <source>Scientia Agricola</source> <volume>82</volume>, <elocation-id>e20240115</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1590/1678-992x-2024-0115</pub-id>
</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Tao</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Green fruit detection with a small dataset under a similar color background based on the improved YOLOv5-AT</article-title>. <source>Foods</source> <volume>13</volume>, <elocation-id>1060</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/foods13071060</pub-id>, PMID: <pub-id pub-id-type="pmid">38611366</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hou</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Inversion study of nitrogen content of hyperspectral apple canopy leaves using optimized least squares support vector machine approach</article-title>. <source>Forests</source> <volume>15</volume>, <elocation-id>268</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/f15020268</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>L. Q.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>W. B.</given-names>
</name>
<name>
<surname>Liew</surname> <given-names>A. W. C.</given-names>
</name>
<name>
<surname>You</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An evidential combination method with multi-color spaces for remote sensing image scene classification</article-title>. <source>Inf. Fusion</source> <volume>93</volume>, <fpage>209</fpage>&#x2013;<lpage>226</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inffus.2022.12.025</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Quan</surname> <given-names>L. Z.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>G. Y.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Geng</surname> <given-names>T. Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A conceptual evaluation of a weed control method with post-damage application of herbicides: A composite intelligent intra-row weeding robot</article-title>. <source>Soil Tillage Res.</source> <volume>234</volume>, <elocation-id>105837</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.still.2023.105837</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ling</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Accurate recognition of jujube tree trunks based on contrast limited adaptive histogram equalization image enhancement and improved YOLOv8</article-title>. <source>Forests</source> <volume>15</volume>, <elocation-id>625</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/f15040625</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Men</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>D.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>a). <article-title>Research on multi-cluster green persimmon detection method based on improved Faster RCNN</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>, <elocation-id>1177114</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2023.1177114</pub-id>, PMID: <pub-id pub-id-type="pmid">37346117</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>X. H.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>G. H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>b). <article-title>A no-reference panoramic image quality assessment with hierarchical perception and color features</article-title>. <source>J. Visual Communication Image Representation</source> <volume>95</volume>, <elocation-id>103885</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jvcir.2023.103885</pub-id>
</citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lv</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>L. M.</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>Y. W.</given-names>
</name>
<name>
<surname>Rong</surname> <given-names>H. L.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An image rendering-based identification method for apples with different growth forms</article-title>. <source>Comput. Electron. Agric.</source> <volume>211</volume>, <elocation-id>108040</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108040</pub-id>
</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mao</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>A scalable multi-modal learning fruit detection algorithm for dynamic environments</article-title>. <source>Front. Neurorobotics</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fnbot.2024.1518878</pub-id>, PMID: <pub-id pub-id-type="pmid">39980656</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname> <given-names>S. Y.</given-names>
</name>
<name>
<surname>Ahamed</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Pear recognition in an orchard from 3D stereo camera datasets to develop a fruit picking mechanism using mask R-CNN</article-title>. <source>Sensors</source> <volume>22</volume>, <elocation-id>4187</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s22114187</pub-id>, PMID: <pub-id pub-id-type="pmid">35684807</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>H. B.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y. N.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y. F.</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Recognition method of oil tea fruit in natural scene based on YOLO v5s</article-title>. <source>J. Agric. Machinery</source> <volume>53</volume>, <fpage>234</fpage>&#x2013;<lpage>242</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.6041/j.issn.1000-1298.2022.07.024</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname> <given-names>Y. N.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S. H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>G. D.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>Z. Z.</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>MD-YOLO: Multi-scale Dense YOLO for small target pest detection</article-title>. <source>Comput. Electron. Agric.</source> <volume>213</volume>, <elocation-id>108233</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108233</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ullah</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Panzarov&#xe1;</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Trt&#xed;lek</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lexa</surname> <given-names>M.</given-names>
</name>
<name>
<surname>M&#xe1;&#x10d;ala</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Neumann</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>High-throughput spike detection in greenhouse cultivated grain crops with attention mechanisms-based deep learning models</article-title>. <source>Plant Phenomics</source> <volume>6</volume>, <elocation-id>155</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/plantphenomics.0155</pub-id>, PMID: <pub-id pub-id-type="pmid">38476818</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Design of and experimentation on an intelligent intra-row obstacle avoidance and weeding machine for orchards</article-title>. <source>Agriculture</source>, <fpage>15</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture15090947</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y. H.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An automated learning method of semantic segmentation for train autonomous driving environment understanding</article-title>. <source>IEEE Trans. Ind. Inf.</source> <volume>20</volume>, <fpage>6913</fpage>&#x2013;<lpage>6922</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TII.2024.3353874</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>P. H.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>F. S.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S. Q.</given-names>
</name>
<name>
<surname>Ning</surname> <given-names>J. F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Visual recognition of cherry tomatoes in plant factory based on improved deep instance segmentation</article-title>. <source>Comput. Electron. Agric.</source> <volume>197</volume>, <elocation-id>106991</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.106991</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>F. Z.</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>X. Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z. J.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Center Net-based fast recognition method for multi-apple targets in dense scenes</article-title>. <source>J. Agric. Machinery</source> <volume>53</volume>, <fpage>265</fpage>&#x2013;<lpage>273</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.6041/j.issn.1000-298.2022.02.028</pub-id>
</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>W. H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>W. M.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W. C.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X. Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Remote sensing image instance segmentation network with transformer and multi-scale feature representation</article-title>. <source>Expert Syst. Appl.</source> <volume>234</volume>, <elocation-id>121007</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2023.121007</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yi</surname> <given-names>W. G.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Research on Underwater small target Detection Algorithm based on improved YOLOv7</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>66818</fpage>&#x2013;<lpage>66827</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2023.3290903</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhuang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A dichotomy color quantization algrithm for the HSI color space</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>8135</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-023-34977-0</pub-id>, PMID: <pub-id pub-id-type="pmid">37208419</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Three-dimensional pose detection method based on keypoints detection network for tomato bunch</article-title>. <source>Comput. Electron. Agric.</source> <volume>195</volume>, <elocation-id>106824</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.106824</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L. Y.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>J. H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Augmented FCN: rethinking context modeling for semantic segmentation</article-title>. <source>Sci. China Inf. Sci.</source> <volume>66</volume>, <fpage>142105</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11432-021-3590-1</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhenyu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Hanjie</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Yuanyuan</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Changyuan</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xiu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Research on an orchard row centreline multipoint autonomous navigation method based on LiDAR</article-title>. <source>Artif. Intell. Agric.</source> <volume>15</volume> (<issue>2</issue>). doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aiia.2024.12.003</pub-id>
</citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>X. C.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y. X.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>W. J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>H. J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Cellular binary neural network for accurate image classification and semantic segmentation</article-title>. <source>IEEE Trans. Multimedia</source> <volume>25</volume>, <fpage>8064</fpage>&#x2013;<lpage>8075</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TMM.2022.3233255</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>W. J.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>J. S.</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>CCAFNet: Crossflow and cross-scale adaptive fusion network for detecting salient objects in RGB-D images</article-title>. <source>IEEE Trans. Multimedia</source> <volume>24</volume>, <fpage>2192</fpage>&#x2013;<lpage>2204</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TMM.2021.3077767</pub-id>
</citation></ref>
</ref-list>
</back>
</article>