<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1533206</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Recognition and localization of ratoon rice rolled stubble rows based on monocular vision and model fusion</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Li</surname>
<given-names>Yuanrui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2901702"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Xiao</surname>
<given-names>Liping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Zhaopeng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Muhua</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Fang</surname>
<given-names>Peng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2911371"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Xiongfei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Jiajia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lin</surname>
<given-names>Jinlong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2291699"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cai</surname>
<given-names>Jinping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Engineering, Jiangxi Agricultural University</institution>, <addr-line>Nanchang</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Jiangxi Key Laboratory of Modern Agricultural Equipment</institution>, <addr-line>Nanchang</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Huajian Liu, University of Adelaide, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: He Jie, South China Agricultural University, China</p>
<p>Enze Duan, Jiangsu Academy of Agricultural Sciences (JAAS), China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Peng Fang, <email xlink:href="mailto:fangpeng@jxau.edu.cn">fangpeng@jxau.edu.cn</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>31</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1533206</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>01</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Li, Xiao, Liu, Liu, Fang, Chen, Yu, Lin and Cai</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Li, Xiao, Liu, Liu, Fang, Chen, Yu, Lin and Cai</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Ratoon rice, as a high-efficiency rice cultivation mode, is widely applied around the world. Mechanical righting of rolled rice stubble can significantly improve yield in regeneration season, but lack of automation has become an important factor restricting its further promotion.</p>
</sec>
<sec>
<title>Methods</title>
<p>In order to realize automatic navigation of the righting machine, a method of fusing an instance segmentation model and a monocular depth prediction model was used to realize monocular localization of the rolled rice stubble rows in this study.</p>
</sec>
<sec>
<title>Results</title>
<p>To achieve monocular depth prediction, a depth estimation model was trained on training set we made, and absolute relative error of trained model on validation set was only 7.2%. To address the problem of degradation of model's performance when migrated to other monocular cameras, based on the law of the input image&#x2019;s influence on model's output results, two optimization methods of adjusting inputs and outputs were used that decreased the absolute relative error from 91.9% to 8.8%. After that, we carried out model fusion experiments, which showed that CD (chamfer distance) between predicted 3D coordinates of navigation points obtained by fusing the results of the two models and labels was only 0.0990. The CD between predicted point cloud of rolled rice stubble rows and label was only 0.0174.</p>
</sec>
</abstract>
<kwd-group>
<kwd>ratoon rice</kwd>
<kwd>model fusion</kwd>
<kwd>depth prediction</kwd>
<kwd>deep learning</kwd>
<kwd>monocular vision</kwd>
</kwd-group>
<contract-sponsor id="cn001">Foundation of Jiangxi Educational Commission<named-content content-type="fundref-id">10.13039/501100019227</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Jiangxi Provincial Department of Science and Technology<named-content content-type="fundref-id">10.13039/501100010857</named-content>
</contract-sponsor>
<counts>
<fig-count count="11"/>
<table-count count="5"/>
<equation-count count="7"/>
<ref-count count="33"/>
<page-count count="14"/>
<word-count count="6407"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Technical Advances in Plant Science</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Ratoon rice is a rice cultivation method that is planted once and harvested twice (<xref ref-type="bibr" rid="B11">Firouzi et&#xa0;al., 2018</xref>), which has the advantages of short fertility, high yield, low cost and sustainable economic benefits compared to ordinary rice (<xref ref-type="bibr" rid="B28">Yang et&#xa0;al., 2024</xref>). Ratoon rice is grown in many parts of the world, mainly in East and South Asia, some countries in Africa, southern United States and Latin America (<xref ref-type="bibr" rid="B24">Pasaribu et&#xa0;al., 2018</xref>). However, existing harvesters often cause a large area of rolling damage when harvesting the first season of ratoon rice, leading to a decline in the yield of the regeneration season, which seriously affects yield and restricts further promotion of planting area (<xref ref-type="bibr" rid="B27">Xiao, 2018</xref>). To solve this problem, our research team developed a rolled rice stubble righting machine, which was shown to significantly increase the yield of ratoon rice (<xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B6">Chen et&#xa0;al., 2023</xref>). The righting machine is mounted on the back of a paddy field vehicle, and the driver needs to concentrate highly on observing the relative positions of the back of rolled stubble row and righting machine, so the driver is easily fatigued, and due to the rolled stubble row is very irregular, the righting accuracy needs to be improved. In order to solve the above problems, the righting machine needs to realize the automatic row alignment, and obtaining spatial position of rolled stubble row is a prerequisite.</p>
<p>Currently, Global Navigation Satellite System (GNSS), Light Detection and Ranging (LIDAR), and machine vision are the commonly used sensing methods for obtaining navigation and position information in agriculture, GNSS can only provide absolute position information, which is suitable for use under the condition of fixed position of crop rows (<xref ref-type="bibr" rid="B3">Bonadies and Gadsden, 2019</xref>). LIDAR can obtain the relative distances of the objects in a certain area, but it has the problems of sparse point cloud data, and it is sensitive to rain, fog, dust, etc (<xref ref-type="bibr" rid="B32">Zhang et&#xa0;al., 2022</xref>). Machine vision detects the position of objects by acquiring its color, texture, shape, and other features through vision sensors (<xref ref-type="bibr" rid="B17">Kim et&#xa0;al., 2021</xref>), which is less costly than the previous two methods, and its robustness strengthens with algorithmic enhancement, and is widely used in field navigation operations, such as weeding, tilling, spraying, etc (<xref ref-type="bibr" rid="B31">Zhang et&#xa0;al., 2024</xref>).</p>
<p>In crop row-based machine vision navigation applications, the common method is to first obtain the position of the navigation reference target on the image through image processing algorithms, the commonly used algorithms can be classified into image processing algorithms based on traditional image processing and based on deep learning, the deep learning model benefits from its powerful information extraction ability, even in complex environments, can also obtain high recognition accuracy, such as semantic segmentation model and instance segmentation models and so on (<xref ref-type="bibr" rid="B23">O&#x2019;Mahony et&#xa0;al., 2020</xref>). Then the navigation deviation is determined based on the position of the recognized target and the declination of the forward direction (<xref ref-type="bibr" rid="B30">Yuan et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B18">Kong et&#xa0;al., 2025</xref>). Although this method is straightforward, it cannot acquire the metric relative distance of the navigation target with regard to the working machine, in order to solve this problem, some researchers have used multi-sensor fusion to acquire distance information while acquiring the RGB image, such as binocular camera (fusion of two monocular cameras) (<xref ref-type="bibr" rid="B19">Li et&#xa0;al., 2022</xref>), RGB-D camera (RGB camera fusion of depth sensor) (<xref ref-type="bibr" rid="B26">Silva et&#xa0;al., 2022</xref>), RGB camera fusion LIDAR (<xref ref-type="bibr" rid="B15">HE et&#xa0;al., 2022</xref>), etc., but these methods still have many problems, such as complex fusion algorithms, sparse point clouds, high cost, and complex calibration.</p>
<p>In recent years, with the development of deep learning, the information contained in images has been further mined, in addition to models such as target detection and instance segmentation, some scholars have proposed a monocular depth prediction model, which predicts the depth (distance relative to the camera) at each pixel position from a single image (<xref ref-type="bibr" rid="B1">Abuolaim and Brown, 2020</xref>), even though the training process uses sparse depth maps, the trained model is capable of outputting dense depth maps. In combination with the camera&#x2019;s intrinsics, depth can also be converted into a 3D point cloud in the camera&#x2019;s coordinate system (<xref ref-type="bibr" rid="B12">Fu et&#xa0;al., 2021</xref>). In a recent study, the state-of-the-art model had only 3.9% absolute relative error in predicted depth versus label on the KITTI dataset (<xref ref-type="bibr" rid="B16">Hu et&#xa0;al., 2024</xref>), demonstrating its potential for autonomous driving, robotics, 3D reconstruction, and more. In recent years, depth prediction models have begun to be applied in the field of agriculture. <xref ref-type="bibr" rid="B33">Zhao et&#xa0;al. (2022)</xref> used the P3ES-Net depth prediction model to reconstruct a 3D point cloud of a plant from a single image and measured the phenotypic parameters of the plant from the point cloud. <xref ref-type="bibr" rid="B9">Cui et&#xa0;al. (2022)</xref> used the MonoDA model to achieve monocular depth prediction in a vineyard environment, with an absolute relative error of 13.4%. <xref ref-type="bibr" rid="B8">Coll-Ribes et&#xa0;al. (2023)</xref> improved the accuracy of grapes image instance segmentation by fusing depth and RGB information, where the depth is predicted by a model, improving the F1 value from 0.882 to 0.924 compared to using only RGB images. <xref ref-type="bibr" rid="B25">Shu et&#xa0;al. (2021)</xref> built a field SLAM system using a monocular depth prediction model, which allowed the system to get rid of the LIDAR and stereo cameras and can be easily deployed on existing equipment. Although there has been an influx of research on the application of depth prediction models in agriculture, more application scenarios still need to be explored, such as monocular visual navigation.</p>
<p>In a recent study, we used a deep learning model to achieve instance segmentation of ratoon rice rolled stubble rows (<xref ref-type="bibr" rid="B21">Li et&#xa0;al., 2023</xref>), but in order to achieve automatic navigation, it is also necessary to locate the position of the stubble rows. In this paper, machine vision is used to realize monocular vision-based 3D spatial localization of rolled stubble rows of ratoon rice, unlike the multi-sensor fusion approach, this study adopts a deep learning model fusion-based method, where one model is used for recognition and the other is used for localization, in which the recognition model is an instance segmentation model, which is one of our previous research results (<xref ref-type="bibr" rid="B21">Li et&#xa0;al., 2023</xref>), and the localization model is a depth prediction model. We investigated the depth prediction performance of the depth prediction model under the ratoon rice field scene, and finally fused the outputs of the instance segmentation model and the depth prediction model to obtain the spatial location of the navigation line and 3D point cloud of ratoon rice rolled stubble rows.</p>
<p>The main structure of this paper is as follows: in Section 2, we first introduced the model fusion method, and then described the structure and training method of the depth prediction model used in this paper. In Section 3, we trained the depth prediction model, tested the model performance on a validation set, and then obtained the law of the influence of the change of the focal length of the input image on the depth value predicted by the model, according to which we proposed two optimization methods to improve the performance of the model migrating to a monocular camera. Finally, we conducted model fusion experiments.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Method of recognizing and locating the rolled stubble rows</title>
<p>Flowchart for model fusion is shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, where an image is input into the instance segmentation model and the monocular depth prediction model to output the instance mask and the depth map, respectively. Then, the instance mask is horizontally divided into n blocks, and the average horizontal and vertical coordinates of each instance mask in each block are calculated in the image coordinate system, which are used as the coordinates of the navigation points, and the navigation lines are formed by connecting the navigation points in the instances. According to the image coordinates of the navigation point, corresponding depth value can be obtained in the depth map. Then, the planar 2D image is converted to a 3D point in the camera coordinate system according to a conversion equation between image coordinates and camera coordinates in the camera imaging principle:</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Flowchart of the navigation line localization method based on model fusion. Each color of mask represents a row of rolled rice stubble. The image coordinate system (u, v) is in pixel and the camera coordinate system (X, Y, Z) is in m. The size of the depth map is equal to that of the input image, and the valid depth value (greater than 0) on each pixel indicates the vertical distance of the object from the camera, i.e., the value of the Z axis in the camera coordinate system.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g001.tif"/>
</fig>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mi>u</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>v</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
<mml:mo>[</mml:mo>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mi>X</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>Y</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>Z</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Z equals depth, u, v are image coordinates, and X, Y, Z are 3D coordinates in the camera coordinate system, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are intrinsic parameters that represent the pixel-represented focal length in x and y directions, the x and y coordinates of the optical center in the image coordinate system, respectively.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Depth dataset</title>
<p>The dataset is collected using two depth cameras, one for training and validation, and the other for testing the generalization ability of the trained model. This is critical in real-world deployments, as the trained models will be migrated to the monocular camera for deployment.</p>
<sec id="s2_2_1">
<label>2.2.1</label>
<title>Data collection in the field</title>
<p>The dataset was collected in Cailing Town, Duchang County, Jiujiang City, Jiangxi Province, China, in the mornings of August 10 and 11 2024, under sunny weather. The collection environment was a paddy field after the first harvest of ratoon rice, and the collection environment and equipment is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, two depth cameras were used to collect data, the Intel RealSense D457 (D457) and Stereolabs ZED (ZED), the resolution of the images collected by the D457 is 1280*720, and that of the images collected by the ZED is 640*360, their models are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Figure&#xa0;3 of the Supplementary Material</bold>
</xref>, and setup information is given in <xref ref-type="supplementary-material" rid="SM1">
<bold>Table&#xa0;2 of the Supplementary Material</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Dataset collection scene and collection equipment. The equipment consists of a power supply and controller, a monitor, two depth cameras, and a Yanmar Paddy Chassis. The controller is an AGX Xavier Orin. The images taken by the cameras are transmitted to the monitor in real time. Two depth cameras mounted side by side with lenses tilted to look at the ground.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g002.tif"/>
</fig>
<p>As shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, Two depth cameras were mounted in front of the paddy field vehicle, and the vehicle was manually driven through the field, with the two depth cameras automatically collecting data every 0.2 seconds. As the environments in the centre of the field were very similar and the environments at the edges of the field varied considerably, the vehicle was driven around the boundaries most of the time when collecting data on a field in order to increase the diversity of the data.</p>
</sec>
<sec id="s2_2_2">
<label>2.2.2</label>
<title>Dataset production</title>
<p>In order to improve the reading and writing efficiency of the depth data, the png image format was adopt to store the depth data, the original data collected were of floating type with the unit of m which becomes mm after being multiplied by 1000, and the data type was converted to unsigned 16-bit integer and saved, which can retain the precision of three decimals. 33706 sets of data were collected by the two depth cameras, D457 and ZED, each set of data contains a left eye RGB image and a depth map, 19808 sets collected by the D457 depth camera, and 13898 sets collected by the ZED depth camera. All data can be categorized into general and obstacle according to the scene, several sets of data are shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, where the black part is the missing depth value. The depth values were saved in the range of 0-20 meters.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Dataset visualization. <bold>(A, B)</bold> denote general, obstacle type data captured by the D457 camera, <bold>(C, D)</bold> denote general, obstacle type data captured by the ZED camera, <bold>(E)</bold> Color bar for depth map, m. The first column is the rgb image and the second column is the depth map. The black part of the depth map indicates that the depth information is null here. The unit of the color bar is m.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g003.tif"/>
</fig>
<p>The collected data are divided into training, validation and test sets, where the training set contains 16,808 sets of data from the D457, the other 2,000 sets of data from the D457 are the validation set, and 2000 images captured by ZED were randomly selected as test set to test the performance of the model migrated to the monocular camera.</p>
</sec>
<sec id="s2_2_3">
<label>2.2.3</label>
<title>Dataset augmentation</title>
<p>Data augmentation has proved to be a robust technique for solving a variety of challenging deep learning tasks, including image classification, natural language understanding, speech recognition, and semi-supervised learning (<xref ref-type="bibr" rid="B13">Gong et&#xa0;al., 2021</xref>). The main method (DNNs) to improve the generalization ability of deep neural networks is data augmentation by expanding the training set through data transformation (<xref ref-type="bibr" rid="B10">Dabouei et&#xa0;al., 2021</xref>). In this study, in order to reduce or even eliminate the effect of color during model training and to avoid the depth prediction model predicting depth based on color information, we used image enhancement techniques during model training. Before feeding the images into the model, we performed random color transformations on the images and all the transformation operations are done by Albumentations (<xref ref-type="bibr" rid="B4">Buslaev et&#xa0;al., 2020</xref>).The flow chart of transformation is shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Figure&#xa0;1 of the Supplementary Material</bold>
</xref>, the partial data augmentation results for one image are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Figure&#xa0;2 of the Supplementary Material</bold>
</xref>.</p>
</sec>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Monocular depth prediction model</title>
<sec id="s2_3_1">
<label>2.3.1</label>
<title>Model structure</title>
<p>The depth prediction modeling framework used in this study is BinsFormer (<xref ref-type="bibr" rid="B20">Li et&#xa0;al., 2024</xref>), and its overall structure is shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, which consists of three parts: a pixel-level module, a transformer module, and a depth prediction module. An image is first fed into the backbone network (The swin-L (<xref ref-type="bibr" rid="B22">Liu et&#xa0;al., 2021</xref>) was used) in the pixel-level decoding module, which extracts features from the image and then decodes them into multiscale features F and pixel-level representations. Then, in the Transformer module, queries interact with F with the help of the attention mechanism and their outputs go into the independent MLPs. MLPs output embeddings into bins predictions and bins embeddings. Then the model predicts the probability distribution map via a dot product between pixel representations and bins embeddings in depth estimation module. The final depth estimation is calculated by a linear combination between the probability distribution map and post-processed N bins centers. The model output are absolute depths in m.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>BinsFormer overview.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g004.tif"/>
</fig>
</sec>
<sec id="s2_3_2">
<label>2.3.2</label>
<title>Model training methods</title>
<p>The training process of the depth prediction model is shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. During the training process, in addition to the input RGB images for inference, labels are needed to calculate the loss, the labels include the depth map captured by the depth camera and the corresponding mask, the mask is the part of the label that is not null, the mask avoids missing part (denoted by 0) of label depth map is involved in the loss calculation. The loss function used in training is silog (<xref ref-type="bibr" rid="B2">Bhat et&#xa0;al., 2023</xref>). The data used in training is the D457 training set. The hardware used for training is mainly a Xeon (R) Platinum 8358P CPU, 10 NVIDIA GeForce 4090 24G GPUs. the training hyperparameters are shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. To highlight that the model was trained on the D457 dataset, this model is uniformly referred to as &#x2018;Model-D457&#x2019; in the following.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Flowchart for training of depth prediction model. The white portion of the Mask has a value of 1 and the black portion has a value of 0. 'Prediction*Mask' indicates a pixel-by-pixel multiplication of the depth in the prediction depth map and the mask.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g005.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Hyperparameters for model training.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Hyperparameters</th>
<th valign="middle" align="center">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Optimizer</td>
<td valign="top" align="left">AdamW</td>
</tr>
<tr>
<td valign="top" align="left">Weight decay</td>
<td valign="top" align="left">1.00e-04</td>
</tr>
<tr>
<td valign="top" align="left">Initial learning rate</td>
<td valign="top" align="left">0.001</td>
</tr>
<tr>
<td valign="top" align="left">Max iters</td>
<td valign="top" align="left">40000</td>
</tr>
<tr>
<td valign="top" align="left">Validation interval</td>
<td valign="top" align="left">1000</td>
</tr>
<tr>
<td valign="top" align="left">Minibatch size</td>
<td valign="top" align="left">10</td>
</tr>
<tr>
<td valign="top" align="left">Learning rate scheduler</td>
<td valign="top" align="left">Poly</td>
</tr>
<tr>
<td valign="top" align="left">Learning rate drop power</td>
<td valign="top" align="left">0.5</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Evaluation metrics</title>
<p>In this study, eight commonly used evaluation metrics were used to assess the accuracy performance of the depth prediction model, including absolute relative error (ABS-REL), root mean squared error (RMSE), threshold accuracy (&#x3b4;1, &#x3b4;2, &#x3b4;3) etc., and their calculation methods are as follows.</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtext>ABS-REL</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>&#x2211;</mml:mo>
<mml:mo stretchy="false">|</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">|</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mtext>RMSE</mml:mtext>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtext>Silog</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msup>
<mml:mi>N</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mtext>n</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>%</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>100</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>*</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&lt;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>1.25</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#xa0;&#xa0;</mml:mtext>
<mml:msup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mtext>RMSElog</mml:mtext>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mtext>SQ-REL</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the model prediction at the pixel position and <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the depth value (not null) captured by the depth camera at the corresponding position. N denotes the total number of <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the depth map. All are smaller and better except &#x3b4;n which is larger and better.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and analysis</title>
<p>In this chapter, the depth prediction model was trained and its performance was tested on the validation set, then we obtained the law of the influence of focal length through the model prediction results, based on which two optimizations were proposed to improve the performance of the model migrated to a monocular camera (ZED test set). Once the model was ready, model fusion experiments were carried out to obtain the 3D coordinates of the navigation points and the rolled rice stubble rows, and the CD values between them and the labels were calculated.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Model training process and performance on validation set</title>
<p>The change curves of some important values of Model-D457 during the training process are shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>. As the training proceeds, the loss decreases significantly, and the evaluation metrics &#x3b4;1 rises, RMSE and ABS-REL show a decreasing trend, indicating that the depth information predicted by the model is getting closer and closer to the depth captured by the depth camera. Although the loss still decreases significantly after iteration up to 10000, the accuracy on the validation set does not improve significantly, which may be due to the learning rate being too small at this time. The best RMSE is achieved on the validation set at 17000 training iterations, and the model weights at this point are saved as the final weights.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Change curves of some key values during training. <bold>(A)</bold> loss change curve, <bold>(B)</bold> &#x3b4;1 change curve on validation set, <bold>(C)</bold> RMSE change curve on validation set, <bold>(D)</bold> ABS-REL change curve on validation set, <bold>(E)</bold> Learning rate change curve. Their horizontal directions are all in iter.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g006.tif"/>
</fig>
<p>The visualization of Model-D457&#x2019;s inference results on several images of the validation set is shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. By comparing the full predicted depth maps and labels, it can be seen that despite the large number of nulls in the label depth maps used in the training process, but the predicted depth map has significant depth errors at long range due to the absence of long-distance depth label. Comparison of the label depth map and the predicted depth map at the corresponding locations shows that the depth values obtained by the model inference are very close to the label depth map in color, which is further evidenced by the error maps, where the difference between the label depths and the predicted depths is so small that it presents large areas of white color, in addition, we also find that the larger error values are concentrated at the far distance, which presents blue and red colors in error map.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Visual comparison of Model-D457's output on the validation set with labels. <bold>(A)</bold> Input images, <bold>(B)</bold> Full predicted depth map, <bold>(C)</bold> Predicted depth map at label position, indicates the portion of the predicted depth map that corresponds to valid values at the label position, <bold>(D)</bold> Label depth map, acquired by the D457 camera, <bold>(E)</bold> Error map, is calculated as the valid depth value in the label depth map minus the predicted value pixel by pixel, <bold>(F)</bold> Color bar for error map, m. The color bar for depth map is shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g007.tif"/>
</fig>
<p>In <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, the ABS-REL between the predicted and label values of the model is only 7.2%, the RMSE is only 0.383, and the other evaluation metrics also exhibit small errors. The above results show that the depth prediction value of Model-D457 is very close to the measured value of D457 depth camera, which demonstrates its strong spatial perception ability.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Performance of Model-D457 on the validation set.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">ABS-REL&#x2193;</th>
<th valign="middle" align="center">RMSE&#x2193;</th>
<th valign="middle" align="center">Silog&#x2193;</th>
<th valign="middle" align="center">&#x3b4;1&#x2191;</th>
<th valign="middle" align="center">&#x3b4;2&#x2191;</th>
<th valign="middle" align="center">&#x3b4;3&#x2191;</th>
<th valign="middle" align="center">RMSElog&#x2193;</th>
<th valign="middle" align="center">SQ-REL&#x2193;</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">7.2%</td>
<td valign="middle" align="left">0.383</td>
<td valign="middle" align="left">0.037</td>
<td valign="middle" align="left">0.984</td>
<td valign="middle" align="left">0.999</td>
<td valign="middle" align="left">0.999</td>
<td valign="middle" align="left">0.039</td>
<td valign="middle" align="left">0.024</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Effect of pixel-represented focal length on depth prediction</title>
<p>Pixel-represented focal length expresses the physical focal length in pixels, zooming in on an image causes pixel-represented focal length to increase, zooming out causes it to decrease, and cropping an image does not change pixel-represented focal length. Recently, a scholars have achieved the training of a depth prediction model on mixed dataset by adjusting the pixel-represented focal lengths on different publicly available datasets to be consistent (<xref ref-type="bibr" rid="B29">Yin et&#xa0;al., 2023</xref>), which demonstrated that the pixel-represented focal length of the input image has an effect on the prediction results of the model, but the exact effect is not clear, and we experimentally explore the exact effect in this section.</p>
<p>We carried out experiments using Model-D457 on one of the images in the validation set and obtained its outputs under four conditions, which are visualized in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, and the evaluation metrics are calculated in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>:</p>
<list list-type="order">
<list-item>
<p>In the results of the original image, both the visualization results and the evaluation metrics calculations show that the predicted and label values of the model are close;</p>
</list-item>
<list-item>
<p>After the original image is centrally cropped to 360*640 from 720*1280 resolution, the predicted and label values are also close to each other, and the RMSE is slightly increased compared to that before the cropping, and the error map is reddish at distance and bluish in near area before cropping, and blueish at distance and reddish in near area after cropping, the reason for this phenomenon is still unknown;</p>
</list-item>
<list-item>
<p>After the cropped image is enlarged to 720*1280, there are obvious color differences between the predicted depth map and the label depth map, and the error map as a whole is reddish, indicating that the predicted depth value of the model is smaller than the label value as a whole, and comparing the calculation results of the evaluation metrics before and after the enlargement, the ABS-REL grows from 5% to 39.6%, the RMSE increases from 0.442 to 2.427, and the performance of other evaluation metrics also decreases significantly;</p>
</list-item>
<list-item>
<p>After the original image is reduced to 360*640, there are obvious color differences between the predicted depth map and the label depth map, and the error map as a whole is blueish, indicating that the predicted depth value of the model is greater than the label value as a whole, the performance of the evaluation metrics also decreased significantly.</p>
</list-item>
</list>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Visualization of Model-D457 output results under four transformations on one image of the validation set. Original: the result of the original image; 0.5&#xd7;Crop: the result of the original image after 0.5&#xd7;central cropping; 0.5&#xd7;Crop+2&#xd7;Resize: the result of the original image after 0.5&#xd7;central cropping and 2&#xd7;magnification; 0.5&#xd7;Resize: the result after 0.5x reduction of the original image. <bold>(A)</bold> Input images, <bold>(B)</bold> Full predicted depth map, <bold>(C)</bold> Predicted depth map at label position, indicates the portion of the predicted depth map that corresponds to valid values at the label position, <bold>(D)</bold> Label depth map, acquired by the D457 camera, <bold>(E)</bold> Error map, is calculated as the valid depth value in the label depth map minus the predicted value pixel by pixel, The color bar of the error map is shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. The color bar of depth map is shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g008.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>The performance of Model-D457 under different transformations of an image in the validation set.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">ABS-REL&#x2193;</th>
<th valign="middle" align="center">RMSE&#x2193;</th>
<th valign="middle" align="center">Silog&#x2193;</th>
<th valign="middle" align="center">&#x3b4;1&#x2191;</th>
<th valign="middle" align="center">&#x3b4;2&#x2191;</th>
<th valign="middle" align="center">&#x3b4;3&#x2191;</th>
<th valign="middle" align="center">RMSElog&#x2193;</th>
<th valign="middle" align="center">SQ-REL&#x2193;</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">
<bold>Original</bold>
</td>
<td valign="middle" align="left">5.3%</td>
<td valign="middle" align="left">0.369</td>
<td valign="middle" align="left">0.030</td>
<td valign="middle" align="left">0.997</td>
<td valign="middle" align="left">0.999</td>
<td valign="middle" align="left">0.999</td>
<td valign="middle" align="left">0.030</td>
<td valign="middle" align="left">0.017</td>
</tr>
<tr>
<td valign="middle" align="left">
<bold>0.5&#xd7;Crop</bold>
</td>
<td valign="middle" align="left">5.0%</td>
<td valign="middle" align="left">0.442</td>
<td valign="middle" align="left">0.025</td>
<td valign="middle" align="left">0.999</td>
<td valign="middle" align="left">0.999</td>
<td valign="middle" align="left">0.999</td>
<td valign="middle" align="left">0.026</td>
<td valign="middle" align="left">0.020</td>
</tr>
<tr>
<td valign="middle" align="left">
<bold>0.5&#xd7;Crop+2&#xd7;Resize</bold>
</td>
<td valign="middle" align="left">39.6%</td>
<td valign="middle" align="left">2.427</td>
<td valign="middle" align="left">0.049</td>
<td valign="middle" align="left">0.003</td>
<td valign="middle" align="left">0.289</td>
<td valign="middle" align="left">0.957</td>
<td valign="middle" align="left">0.227</td>
<td valign="middle" align="left">0.786</td>
</tr>
<tr>
<td valign="middle" align="left">
<bold>0.5&#xd7;Resize</bold>
</td>
<td valign="middle" align="left">90.6%</td>
<td valign="middle" align="left">3.522</td>
<td valign="middle" align="left">0.059</td>
<td valign="middle" align="left">0.035</td>
<td valign="middle" align="left">0.059</td>
<td valign="middle" align="left">0.515</td>
<td valign="middle" align="left">0.282</td>
<td valign="middle" align="left">2.674</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In conclusion, changing the pixel-represented focal length in the input image will affect the output of the Model-D457: Increasing the pixel focal length will result in smaller depths predicted by the model, and decreasing the first-estimate focal length will result in larger depths predicted by the model.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Adjusting inputs and outputs to improve model migration performance</title>
<p>Although Model-D457 performs well on the validation set, the goal of this paper is to migrate the model to a monocular camera, which is in line with practical application scenarios, however, there is a big difference in the pixel-represented focal length between the photos taken by the D457 camera and the ZED camera, and the average evaluation metrics of all the images in the test set are shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. when the original (original) ZED images are directly input into the model, it can be seen that compared to the model&#x2019;s results on the D457 validation set, all the metrics are significantly decreased, the ABS-REL is more than 90%, the RMSE is more than 3.2, and the &#x3b4;1 is only 0.004, indicating that the model&#x2019;s performance is significantly decreased. The visualization results are shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, when the original ZED image is input, there is a significant difference in the color of the predicted depth map compared to the label at the same location, and the error value map shows blue color as a whole, indicates that the depth of prediction is greater than the depth of label.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Performance of Model-D457 migrated to the ZED test set under three conditions.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">ABS-REL&#x2193;</th>
<th valign="middle" align="center">RMSE&#x2193;</th>
<th valign="middle" align="center">Silog&#x2193;</th>
<th valign="middle" align="center">&#x3b4;1&#x2191;</th>
<th valign="middle" align="center">&#x3b4;2&#x2191;</th>
<th valign="middle" align="center">&#x3b4;3&#x2191;</th>
<th valign="middle" align="center">RMSElog&#x2193;</th>
<th valign="middle" align="center">SQ-REL&#x2193;</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">
<bold>Original</bold>
</td>
<td valign="top" align="center">91.9%</td>
<td valign="top" align="center">3.270</td>
<td valign="top" align="center">0.045</td>
<td valign="top" align="center">0.004</td>
<td valign="top" align="center">0.038</td>
<td valign="top" align="center">0.637</td>
<td valign="top" align="center">0.283</td>
<td valign="top" align="center">2.722</td>
</tr>
<tr>
<td valign="top" align="center">
<bold>2.14&#xd7;Resize</bold>
</td>
<td valign="top" align="center">8.8%</td>
<td valign="top" align="center">0.397</td>
<td valign="top" align="center">0.047</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.990</td>
<td valign="top" align="center">0.996</td>
<td valign="top" align="center">0.051</td>
<td valign="top" align="center">0.072</td>
</tr>
<tr>
<td valign="top" align="center">
<bold>Output/1.94</bold>
</td>
<td valign="top" align="center">8.8%</td>
<td valign="top" align="center">0.314</td>
<td valign="top" align="center">0.041</td>
<td valign="top" align="center">0.933</td>
<td valign="top" align="center">0.991</td>
<td valign="top" align="center">0.996</td>
<td valign="top" align="center">0.050</td>
<td valign="top" align="center">0.050</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Visual comparison of Model-D457's performance on the ZED test set in three conditions. The three conditions are the original result without any manipulation, the result of resizing the input image, and the result of rescaling the output. <bold>(A)</bold> Input images, <bold>(B)</bold> Full predicted depth map, <bold>(C)</bold> Predicted depth map at label position, indicates the portion of the predicted depth map that corresponds to valid values at the label position, <bold>(D)</bold> Label depth map, acquired by the ZED camera, <bold>(E)</bold> Error map, is calculated as the valid depth value in the label depth map minus the predicted value pixel by pixel, <bold>(F)</bold> Color bar for depth map, m. The color bar for the error map is shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g009.tif"/>
</fig>
<p>According to the experimental results in section 3.2, for the above-mentioned phenomenon that the predicted depth of the model on the test set is too large, we adopt two optimization methods namely, adjusting the input and adjusting the output to improve the performance of the model on ZED camera, and the results of average evaluation metrics are shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, and results of visualization are shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>:</p>
<list list-type="order">
<list-item>
<p>Adjusting the size of the input image. When the input image of the model is enlarged to 2.14 times of the original, that is, i.e., the resolution of the input image is enlarged to 1369*770, the model performance reaches the best, and at this time, all the evaluation metrices such as ABS-REL and RMSE are significantly improved, and they are even close to the performance of the model in the validation set in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, and the predicted depth maps of the model at this time are very close to the labels in terms of color, and the error map overall close to white, all indicating that the predicted depth values are close to the labels.</p>
</list-item>
<list-item>
<p>Adjusting the scale of the output depth values. After reducing the overall depth value by 1.94 times, the evaluation metric reaches the best, which is close to the result obtained by adjusting the input. The predicted depth maps are also very close in color compared to the labels, and the error maps are also close to white overall, indicating that the predicted depth values are close to the labels. The depth range of the labels used in the above experiment is 0-5m, because in practical use only need to obtain the distance information of the near target, plus the depth of the camera in the long distance when the accuracy is poor, this time the calculated accuracy is meaningless.</p>
</list-item>
</list>
<p>The above experimental results show that it is easy to improve the depth prediction performance of Model-D457 after migrating to the ZED camera by simply resizing the input image or scaling the model output depth overall, and its final results are even close to the model&#x2019;s performance on the validation set.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Model fusion-based rolled rice stubble row recognition and localization experiment</title>
<p>We used SMR-RS (<xref ref-type="bibr" rid="B21">Li et&#xa0;al., 2023</xref>) as the instance segmentation model and Model-D457 as the depth prediction model, and randomly selected 1000 RGB images from the ZED test set for model fusion test, of which the experimental results of three images are shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>:</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Visualization of model fusion experiment results. <bold>(A)</bold> Input ZED images, <bold>(B)</bold>Instance segmentation, the red dot in the instance mask indicates the center of the bounding box, <bold>(C)</bold> Predicted depth map at label position, <bold>(D)</bold> Label depth map, acquired by the ZED camera, <bold>(E)</bold> Navigation points in rgb image, only the positions of the navigation points in the mask area that are within 5m of the depth of the label are calculated, <bold>(F)</bold>Navigation points in camera coordinate system, including predicted and label, corresponds to the navigation point in <bold>(E)</bold>. <bold>(G)</bold> Color bar for depth map, m.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g010.tif"/>
</fig>
<p>As shown in panels A, B and C, the images inputs to the two models output the mask of the recognized instances of the rolled rice stubble rows and the depth map, respectively, and the different color masks represent different instances, the depth maps in the figure are obtained after using the adjusted input optimization in <bold>section 3.3</bold>. As shown in panel E, we divided the mask horizontally of the instances into 14 rows, and calculated the average horizontal and average vertical coordinates of each instance mask in each row in the image coordinate system to obtain the image coordinates of the navigation point. Then based on these image coordinates we took out the depth value at the corresponding position from the predicted depth map and the label depth map, next, based on the equation in Section 2.1, we got the 3D coordinates of the navigation point in the camera coordinate system, <xref ref-type="supplementary-material" rid="SM1">
<bold>Table&#xa0;1 of the Supplementary Material</bold>
</xref> (ZED-770*1369) demonstrates the intrinsics used therein. As shown in panel F, we refer to the 3D navigation points calculated from the predicted depth and label depth as &#x2018;predicted&#x2019; and &#x2018;label&#x2019;, respectively, and we compare the label navigation points and predicted navigation points of these three images in the 3D camera coordinate system (from left to right corresponding to the rolled stubble rows in the input image), and it can be observed that their positions are very close to each other in 3D space.</p>
<p>CD (Chamfer Distance) (<xref ref-type="bibr" rid="B14">Hajdu et&#xa0;al., 2012</xref>) is a metric for assessing the similarity between different point clouds and is commonly used in 3D reconstruction, in this paper, the CD values between label and predicted navigation points are calculated, which is implemented using the code provided in (<xref ref-type="bibr" rid="B7">Christian, 2023</xref>). The average CD values of the above 1000 images obtained with (Resize Input) and without (Original) the optimization method are shown in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> (Navigation 3D Points). When the optimization method is not used, the CD value is as high as 4.70, and when the optimization method is used, the CD value is only 0.09, which is caused by the fact that the optimization method reduces the gap between the predicted depth values and the label depth values, and the gap between the 3D points obtained by depth conversion is also reduced accordingly.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Average chamfer distance between predicted and label 3D points.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">Original</th>
<th valign="middle" align="center">Resize Input</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Navigation 3D Points</td>
<td valign="middle" align="center">4.7033</td>
<td valign="middle" align="center">0.0990</td>
</tr>
<tr>
<td valign="middle" align="center">Crop Row 3D Points</td>
<td valign="middle" align="center">1.3542</td>
<td valign="middle" align="center">0.0174</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We also obtained the label depths and corresponding predicted depths within 5 m of the instance mask location of the rolled rice stubble rows in the three images of <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, then, used the above method, their 3D coordinates in the camera coordinate system were computed. They were plotted in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> (the left-to-right point clouds in the single plot correspond to the left-to-right rolled rice stubble rows in the rgb image.), the first row is predicted point clouds and the second row is label, and it can be observed that they are very similar to each other. The average value of CD (Crop Row 3D Points) between the predicted and label point clouds in all 1000 test images mentioned above is calculated in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>, and the CD value is as high as 1.35 without using the optimization method. After using the optimization method in 3.4, the CD value is only 0.017.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Three-dimensional point cloud of rolled rice stubble rows. The first row is the predicted point cloud and the second row is the label. The title of each plot corresponds to the image name in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>. The color of the point cloud varies with the value of the Z coordinate and the color bar is shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1533206-g011.tif"/>
</fig>
<p>The above results show that the model fusion-based recognition and localization method for rolled stubble rows of ratoon rice proposed in this paper can achieve high-precision spatial localization of the navigation line, and only monocular RGB images were used. In addition, the 3D point cloud of rolled stubble rows demonstrates the application of the method in 3D reconstruction.</p>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusions</title>
<p>In this paper, we propose a novel spatial localization method, in which we fuse the outputs of an instance segmentation model and a monocular depth prediction model and successfully achieve monocular vision-based spatial localization of ratoon rice rolled stubble rows. To realize the depth prediction, we trained Model-D457, and the ABS-REL on the validation set is only 7.2%, and the RMSE is only 0.383, demonstrating the depth prediction performance of the proximity sensor. Through experiments, we obtained the pattern of the influence of the change of the input image on the output of Model-D457: enlarging the resolution of the input image makes the model prediction result smaller, and reducing makes the model prediction result larger, accordingly, we proposed two optimization methods of adjusting the input and adjusting the output, which made the ABS-REL of the model migrated to other cameras decrease from 91.9% to 8.8%. Once the depth prediction model was ready, we conducted model fusion experiment, and the CD value between the predicted 3D coordinates of the navigation points and the labels was only 0.0990. The CD value between the predicted and label point cloud of the rolled rice stubble rows was only 0.0174. The above results show that the Model-D457 can predict depth well in ratoon rice field scene and its accuracy is even close to that of a depth sensor, applying the method of fusing the depth prediction model with the instance segmentation model we achieved localization performance close to sensor fusion (rgb fusion depth sensor), but our method is much cheaper and easier to be deployed on existing devices.</p>
<p>Nevertheless, our method still has some limitations. Limited by the depth measurement accuracy of the depth camera, the accuracy of the predicted depth by the depth prediction model trained in this paper at a long distance (more than 5m) needs to be improved. In addition, the inference speed of the depth prediction model is slow, and the inference time for a single image is about twice as long as that of the instance segmentation model. Based on these deficiencies, in the future, we will focus on making depth datasets with higher accuracy and at longer distances, and researching lightweight depth prediction models or end-to-end models that integrate both depth prediction and instance segmentation, and ultimately applying this method to automate rice stubble righting machines.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>, further inquiries can be directed to the corresponding author/s. The code used in this paper is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/Yuan-rui-Li/MonoDepth.git">https://github.com/Yuan-rui-Li/MonoDepth.git</ext-link>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>YL: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. LX: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; review &amp; editing. ZL: Conceptualization, Data curation, Methodology, Supervision, Writing &#x2013; review &amp; editing. ML: Conceptualization, Investigation, Software, Writing &#x2013; review &amp; editing. PF: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; review &amp; editing. XC: Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Writing &#x2013; review &amp; editing. JY: Data curation, Methodology, Supervision, Validation, Writing &#x2013; review &amp; editing. JL: Conceptualization, Investigation, Software, Supervision, Writing &#x2013; review &amp; editing. JC: Resources, Supervision, Validation, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was supported by the Jiangxi Provincial Key Research and Development Program, Grant Number 20232BBF60015; Science and Technology Research Project of Jiangxi Educational Committee, grant number GJJ2200415.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We thank Pan Feng for help with model training, and Weijian Fang for help with training data collection.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2025.1533206/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2025.1533206/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Abuolaim</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>M. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Defocus deblurring using dual-pixel data</article-title>. In <conf-name>Computer Vision &#x2013; ECCV 2020</conf-name>; <person-group person-group-type="author">
<name>
<surname>Vedaldi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Bischof</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Brox</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Frahm</surname> <given-names>J.-M</given-names>
</name>
</person-group>., Eds. (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <page-range>111&#x2013;126</page-range>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhat</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Birkl</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Wofk</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Wonka</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Muller</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>ZoeDepth: zero-shot transfer by combining relative and metric depth</article-title>. <source>ArXiv</source>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bonadies</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gadsden</surname> <given-names>S. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An overview of autonomous crop row navigation strategies for unmanned ground vehicles</article-title>. <source>Eng. Agriculture Environ. Food</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eaef.2018.09.001</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buslaev</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Iglovikov</surname> <given-names>V. I.</given-names>
</name>
<name>
<surname>Khvedchenya</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Parinov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Druzhinin</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kalinin</surname> <given-names>A. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Albumentations: fast and flexible image augmentations</article-title>. <source>Information</source> <volume>11</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/info11020125</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Design and experiment of finger-chain grain lifter for ratoon rice stubble rolled by mechanical harvesting</article-title>. <source>Inmateh Agric. Eng.</source> <volume>1</volume>, <fpage>361</fpage>&#x2013;<lpage>372</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.35633/INMATEH</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Mo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Mechanical stubble righting after the mechanical harvest of primary rice improves the grain yield of ratooning rice</article-title>. <source>Agronomy</source> <volume>13</volume>, <fpage>2419</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agronomy13092419</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Christian</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). Available online at: <uri xlink:href="https://github.com/chrdiller/pyTorchChamferDistance.git">https://github.com/chrdiller/pyTorchChamferDistance.git</uri> (Accessed <access-date>Octobe 30, 2024</access-date>).</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Coll-Ribes</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Torres-Rodr&#xed;guez</surname> <given-names>I. J.</given-names>
</name>
<name>
<surname>Grau</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Guerra</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Sanfeliu</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Accurate detection and depth estimation of table grapes and peduncles for robot harvesting, combining monocular depth estimation and CNN methods</article-title>. <source>Comput. Electron. Agric.</source> <volume>215</volume>, <elocation-id>108362</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108362</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cui</surname> <given-names>X.-Z.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.-Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.-H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Monocular depth estimation with self-supervised learning for vineyard unmanned agricultural vehicle</article-title>. <source>Sensors</source> <volume>22</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s22030721</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dabouei</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Soleymani</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Taherkhani</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Nasrabadi</surname> <given-names>N. M.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>SuperMix: supervising the mixing data augmentation</article-title>,&#x201d; in <conf-name>2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>13789</fpage>&#x2013;<lpage>13798</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01358</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Firouzi</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Nikkhah</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Aminpanah</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Rice single cropping or ratooning agro-system: which one is more environment-friendly</article-title>? <source>Environ. Sci. pollut. Res.</source> <volume>25</volume>, <fpage>32246</fpage>&#x2013;<lpage>32256</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11356-018-3076-x</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>He</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Single image 3D object reconstruction based on deep learning: A review</article-title>. <source>Multimedia Tools Appl.</source> <volume>80</volume>, <fpage>463</fpage>&#x2013;<lpage>498</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-020-09722-8</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gong</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Chandra</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>KeepAugment: A simple information-preserving data augmentation approach</article-title>,&#x201d; in <conf-name>2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>1055</fpage>&#x2013;<lpage>1064</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.00111</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hajdu</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hajdu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Tijdeman</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Approximation of the euclidean distance by chamfer distances</article-title>. <source>Acta Cybern</source> <volume>20</volume>, <fpage>399</fpage>&#x2013;<lpage>417</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.14232/actacyb.20.3.2012.3</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>HE</surname> <given-names>J.</given-names>
</name>
<name>
<surname>HE</surname> <given-names>J.</given-names>
</name>
<name>
<surname>LUO</surname> <given-names>X.</given-names>
</name>
<name>
<surname>LI</surname> <given-names>W.</given-names>
</name>
<name>
<surname>MAN</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>FENG</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Rice row recognition and navigation control based on multi-sensor fusion</article-title>. <source>Trans. Chin. Soc. Agric. Machinery</source> <volume>53</volume>, <fpage>18</fpage>&#x2013;<lpage>26</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Metric3D v2: A versatile monocular geometric foundation model for zero-shot metric depth and surface normal estimation</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>46</volume>, <fpage>10579</fpage>&#x2013;<lpage>10596</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2024.3444912</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>W.-S.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>D.-H.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>Y.-J.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>W.-S.</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>C.-H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Stereo-vision-based crop height estimation for agricultural robots</article-title>. <source>Comput. Electron. Agric.</source> <volume>181</volume>, <elocation-id>105937</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105937</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kong</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xue</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>A method&#xa0;for&#xa0;recognizing inter-row navigation lines of rice heading stage based on improved ENet network</article-title>. <source>Measurement</source> <volume>241</volume>, <elocation-id>115677</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.measurement.2024.115677</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lloyd</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ward</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Cox</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Coutts</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Fox</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Robotic crop row tracking around weeds using cereal-specific features</article-title>. <source>Comput. Electron. Agric.</source> <volume>197</volume>, <elocation-id>106941</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.106941</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>BinsFormer: revisiting adaptive bins for monocular depth estimation</article-title>. <source>IEEE Trans. Image Process.</source> <volume>33</volume>, <fpage>3964</fpage>&#x2013;<lpage>3976</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2024.3416065</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>SMR-RS: an improved mask R-CNN specialized for rolled rice stubble row segmentation</article-title>. <source>Appl. Sci.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app13169136</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). &#x201c;<article-title>Swin transformer: hierarchical vision transformer using shifted windows</article-title>,&#x201d; in <conf-name>2021 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>. <fpage>9992</fpage>&#x2013;<lpage>10002</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>O&#x2019;Mahony</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Campbell</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Carvalho</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Harapanahalli</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hernandez</surname> <given-names>G. V.</given-names>
</name>
<name>
<surname>Krpalkova</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). &#x201c;<article-title>Deep learning vs. Traditional computer vision</article-title>,&#x201d; in <source>Advances in Computer Vision</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Arai</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Kapoor</surname> <given-names>S.</given-names>
</name>
</person-group> (<publisher-name>Springer International Publishing</publisher-name>, <publisher-loc>Cham</publisher-loc>), <fpage>128</fpage>&#x2013;<lpage>144</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pasaribu</surname> <given-names>P. O.</given-names>
</name>
<name>
<surname>Triadiati</surname>
</name>
<name>
<surname>Anas</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Rice ratooning using the salibu system and the system of rice intensification method influenced by physiological traits</article-title>. <source>Pertanika J. Trop. Agric. Sci.</source> <volume>41</volume>, <fpage>637</fpage>&#x2013;<lpage>654</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Shu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Lesur</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Pagani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Stricker</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>SLAM in the field: an evaluation of monocular mapping and localization on challenging dynamic agricultural environment</article-title>,&#x201d; in <conf-name>2021 IEEE Winter Conference on Applications of Computer Vision (WACV)</conf-name>. <fpage>1760</fpage>&#x2013;<lpage>1770</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/WACV48630.2021.00180</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Silva</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Cielniak</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep learning-based crop row detection for infield navigation of agri-robots</article-title>. <source>J. Field Robotics</source> <volume>41</volume>, <fpage>2299</fpage>&#x2013;<lpage>2321</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Effect of mechanical harvesting of main crop on the grain yield and quality of ratoon crop in ratooned rice</source>. <publisher-name>Hua Zhong Agriculture University</publisher-name>, <publisher-loc>Wuhan, China</publisher-loc>. Master&#x2019;s Thesis.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Mo</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Appropriate stubble height can effectively improve the rice quality of ratoon rice</article-title>. <source>Foods</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/foods13091392</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yin</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). &#x201c;<article-title>Metric3D: towards zero-shot metric 3D prediction from A single image</article-title>,&#x201d; in <conf-name>2023 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>. <fpage>9009</fpage>&#x2013;<lpage>9019</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV51070.2023.00830</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Development of autonomous navigation system based on neural network and visual servoing for row-crop tracking in vegetable greenhouses</article-title>. <source>Smart Agric. Technol.</source> <volume>9</volume>, <elocation-id>100572</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.atech.2024.100572</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>A review of vision-based crop row detection method: focusing on field ground autonomous navigation&#xa0;operations</article-title>. <source>Comput. Electron. Agric.</source> <volume>222</volume>, <elocation-id>109086</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109086</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Cut-edge detection method for wheat harvesting based on stereo vision</article-title>. <source>Comput. Electron. Agric.</source> <volume>197</volume>, <fpage>106910</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.106910</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Phenotypic parameters estimation of plants using deep learning-based 3-D reconstruction from single RGB image</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/LGRS.2022.3198850</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>