<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2021.732968</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multi-Modal Deep Learning for Weeds Detection in Wheat Field Based on RGB-D Images</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Ke</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1333327/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhu</surname> <given-names>Yan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/446195/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Cao</surname> <given-names>Weixing</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1019571/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Jiang</surname> <given-names>Xiaoping</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Jiang</surname> <given-names>Zhijian</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Shuailong</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Ni</surname> <given-names>Jun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1466052/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Agriculture, Nanjing Agricultural University</institution>, <addr-line>Nanjing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>National Engineering and Technology Center for Information Agriculture</institution>, <addr-line>Nanjing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Engineering Research Center of Smart Agriculture, Ministry of Education</institution>, <addr-line>Nanjing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Jiangsu Key Laboratory for Information Agriculture</institution>, <addr-line>Nanjing</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Jiangsu Collaborative Innovation Center for the Technology and Application of Internet of Things</institution>, <addr-line>Nanjing</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>College of Artificial Intelligence, Nanjing Agricultural University</institution>, <addr-line>Nanjing</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Yiannis Ampatzidis, University of Florida, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Liujun Li, Missouri University of Science and Technology, United States; Saeed Hamood Alsamhi, Ibb University, Yemen; Vinay Vijayakumar, University of Florida, United States</p></fn>
<corresp id="c001">&#x002A;Correspondence: Jun Ni, <email>nijun@njau.edu.cn</email></corresp>
<fn fn-type="other" id="fn004"><p>This article was submitted to Sustainable and Intelligent Phytoprotection, a section of the journal Frontiers in Plant Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>11</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>732968</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>06</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>10</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2021 Xu, Zhu, Cao, Jiang, Jiang, Li and Ni.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Xu, Zhu, Cao, Jiang, Jiang, Li and Ni</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Single-modal images carry limited information for features representation, and RGB images fail to detect grass weeds in wheat fields because of their similarity to wheat in shape. We propose a framework based on multi-modal information fusion for accurate detection of weeds in wheat fields in a natural environment, overcoming the limitation of single modality in weeds detection. Firstly, we recode the single-channel depth image into a new three-channel image like the structure of RGB image, which is suitable for feature extraction of convolutional neural network (CNN). Secondly, the multi-scale object detection is realized by fusing the feature maps output by different convolutional layers. The three-channel network structure is designed to take into account the independence of RGB and depth information, respectively, and the complementarity of multi-modal information, and the integrated learning is carried out by weight allocation at the decision level to realize the effective fusion of multi-modal information. The experimental results show that compared with the weed detection method based on RGB image, the accuracy of our method is significantly improved. Experiments with integrated learning shows that mean average precision (<italic>mAP</italic>) of 36.1% for grass weeds and 42.9% for broad-leaf weeds, and the overall detection precision, as indicated by intersection over ground truth (<italic>IoG</italic>), is 89.3%, with weights of RGB and depth images at &#x03B1; = 0.4 and &#x03B2; = 0.3. The results suggest that our methods can accurately detect the dominant species of weeds in wheat fields, and that multi-modal fusion can effectively improve object detection performance.</p>
</abstract>
<kwd-group>
<kwd>weeds detection</kwd>
<kwd>RGB-D image</kwd>
<kwd>multi-modal deep learning</kwd>
<kwd>machine learning</kwd>
<kwd>three-channel network</kwd>
</kwd-group>
<contract-num rid="cn001">2017YFD0201501</contract-num>
<contract-num rid="cn002">XYDXX-049</contract-num>
<contract-num rid="cn003">BE2017385</contract-num>
<contract-num rid="cn003">BE2018399</contract-num>
<contract-num rid="cn003">BE2019306</contract-num>
<contract-sponsor id="cn001">National Key Research and Development Program of China<named-content content-type="fundref-id">10.13039/501100012166</named-content></contract-sponsor>
<contract-sponsor id="cn002">Six Talent Peaks Project in Jiangsu Province<named-content content-type="fundref-id">10.13039/501100010014</named-content></contract-sponsor>
<contract-sponsor id="cn003">Jiangsu Provincial Key Research and Development Program<named-content content-type="fundref-id">10.13039/501100013058</named-content></contract-sponsor>
<counts>
<fig-count count="8"/>
<table-count count="3"/>
<equation-count count="9"/>
<ref-count count="48"/>
<page-count count="10"/>
<word-count count="7250"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="S1">
<title>Introduction</title>
<p>Weeds are a major biological problem that limits the yield and quality of wheat by competing for light, water, fertilizer, and space (<xref ref-type="bibr" rid="B28">Munier-Jolain et al., 2013</xref>; <xref ref-type="bibr" rid="B10">Fahad et al., 2015</xref>). There are both grass and broad-leaf weeds (<xref ref-type="bibr" rid="B12">Gaba et al., 2010</xref>). Grass weeds have narrow, long leaves very similar to those of wheat, and consist mainly of <italic>Echinochloa crusgalli</italic>, <italic>Avena fatua</italic>, and <italic>Aegilops tauschii</italic>. Broad-leaf weeds look different, and include <italic>Pharbitis nil</italic>, <italic>Galium spurium</italic>, <italic>Veronica didyma</italic>, <italic>Capsella bursa-pastoris</italic>, and <italic>Convolvulus arvensis</italic>. Changes in farming practice and the introduction of new wheat varieties have led to significant changes in the species and occurrence of weeds. Grass weeds have invaded and dominated wheat fields, and like broad-leaf weeds, they threaten production (<xref ref-type="bibr" rid="B43">Ulber et al., 2009</xref>). They diminish wheat grain filling and have a greater impact on growth and yield (<xref ref-type="bibr" rid="B39">Siddiqui et al., 2010</xref>). Grass weeds have morphological characteristics and living habits similar to those of wheat, which interfere with their recognition.</p>
<p>Chemical herbicides have become the primary means of farmland weeds management worldwide because of their high efficiency (<xref ref-type="bibr" rid="B24">Jaime and Ricardo, 2017</xref>; <xref ref-type="bibr" rid="B25">Kniss, 2017</xref>). Due to the lack of information on weeds species and distribution, they are sprayed over large areas, resulting in overuse, low utilization, and serious pollution. Although herbicides can directly kill object wild plants, excessive use will cause serious environmental pollution (<xref ref-type="bibr" rid="B36">Rose et al., 2016</xref>), decrease the yield and quality of agricultural products, and reduce the efficiency of agricultural production. Site-specific weeds management (SSWM) is an important solution to herbicide overuse, whose study includes the aspects of crop and weeds detection systems, decision-making algorithms for herbicide application, and weeds control implementation (<xref ref-type="bibr" rid="B7">Camille et al., 2008</xref>; <xref ref-type="bibr" rid="B8">Christensen et al., 2010</xref>), among which the recognition and localization of weeds in fields are key issues.</p>
<p>Since images with high spatial resolution are usually easily available and not costly, they are favored in the integration and application of precision weeds management and adjustable spraying systems. Hence, weeds detection based on digital images is a key technical tool for the accurate recognition and localization of weeds in farmlands (<xref ref-type="bibr" rid="B4">Bakhshipour et al., 2017</xref>). Weeds detection with wheat field images using traditional machine learning methods usually requires the selection of an object area with a sliding window. Manually designed features, such as the color, location, morphology, and texture of wheat and weeds, are analyzed and extracted from wheat field images (<xref ref-type="bibr" rid="B41">Tellaeche et al., 2008</xref>; <xref ref-type="bibr" rid="B31">Petra et al., 2018</xref>). This process fails in multi-scale object detection tasks, the time complexity is high, and manually designed features are sensitive to sample variation (<xref ref-type="bibr" rid="B32">Pflanz et al., 2018</xref>; <xref ref-type="bibr" rid="B45">Xu et al., 2020a</xref>). Deep learning approaches based on CNNs avoid the use of manually designed features, can well detect objects of different sizes, and have been used in the study of weeds detection to significantly improve recognition efficiency (<xref ref-type="bibr" rid="B30">Patr&#x00ED;cioa and Riederb, 2018</xref>; <xref ref-type="bibr" rid="B40">Smith et al., 2019</xref>; <xref ref-type="bibr" rid="B2">Alsamhi et al., 2021</xref>; <xref ref-type="bibr" rid="B37">Saleh et al., 2021</xref>). However, most weeds detection algorithms are based on the input of single-modal images (RGB), and the limited information makes it difficult to recognize different weeds species (<xref ref-type="bibr" rid="B1">Alessandro et al., 2017</xref>; <xref ref-type="bibr" rid="B3">Bah et al., 2018</xref>; <xref ref-type="bibr" rid="B23">Huang et al., 2018b</xref>). In particular, weeds detection in wheat fields is seriously restricted by the similar shapes of grass weeds and wheat, and the lack of recognizability in RGB images.</p>
<p>Studies have demonstrated that the fusion of multi-modal data can effectively improve the robustness of object detection in unfavorable environments, and the combination of RGB, depth and infrared information provides more richer feature space, which facilitates precise classification and detection (<xref ref-type="bibr" rid="B20">Haque et al., 2020</xref>). Since different modalities represent the same scene in different ways, their independence and complementarity can be used to improve the precision of object detection. RGB images, which contain color and texture information, and depth images, which contain geometric structure information, are widely used in object detection tasks because of their high complementarity (<xref ref-type="bibr" rid="B34">Qi et al., 2018</xref>). The pixel value in a depth image reflects the distance between the object and the sensor, and effectively describes the geometric characteristics of the object surface in an image. A depth image is therefore an effective supplement to an RGB image. Plant height is an important feature of growth status, which differs greatly between crops and weeds due to growth competition (<xref ref-type="bibr" rid="B33">Piron et al., 2009</xref>; <xref ref-type="bibr" rid="B48">Zhang and Grift, 2012</xref>). Therefore, we fuse the multi-modal information from RGB and depth images of wheat and weeds to detect weeds in wheat fields. Our work can be summarized as follows:</p>
<list list-type="simple">
<list-item>
<label>(1)</label>
<p>A weeds-in-wheat-field RGB-D dataset for object detection is proposed, including 1,228 RGB images and corresponding depth images, and the weed areas are labeled as broad-leaf and grass.</p>
</list-item>
<list-item>
<label>(2)</label>
<p>To address the CNN&#x2019;s inability to extract abundant features from single-channel depth images, they are recoded by simulating the RGB image structure to generate new structure images that contain more geometric information and are more suitable for CNN-based feature learning.</p>
</list-item>
<list-item>
<label>(3)</label>
<p>According to the concept of multiscale object detection and considering the independence and complementarity of multi-modal data, a three-channel network for weeds detection is proposed from the perspectives of feature-level fusion and decision-level fusion.</p>
</list-item>
</list>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Experimental Design</title>
<p>Experiments on wheat and weeds were carried out from December 2017 to April 2020 at the demonstration base of the National Engineering and Technology Center for Information Agriculture in Rugao County, Nantong City, Jiangsu Province, China. The experimental area was 50 m long and 12 m wide (<xref ref-type="fig" rid="F1">Figure 1</xref>). Weeds were not controlled during field management, and seeds of six weeds species commonly associated with wheat were randomly sown to simulate weeds growth in the open field. <italic>Alopecurus aequalis</italic>, <italic>Poa annua</italic>, <italic>Bromus japonicus</italic>, and <italic>E</italic>. <italic>crusgalli</italic> are grass weeds; <italic>Amaranthus retroflexus</italic> and <italic>C</italic>. <italic>bursa-pastoris</italic> are broad-leaf weeds; and the species composition was similar to that of actual weeds species in wheat fields.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p><bold>(A)</bold> Experimental site; <bold>(B)</bold> Images of all plots; <bold>(C)</bold> RGB image of wheat field. The red star represents the location of our experiments in the map.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g001.tif"/>
</fig>
</sec>
<sec id="S2.SS2">
<title>Image Acquisition and Preprocessing</title>
<p>Wheat field images were acquired using an Intel RealSense Depth Camera D415 (99 mm &#x00D7; 20 mm &#x00D7; 23 mm), an RGB-D camera that adopts active infrared stereo vision technology. As shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>, there were two infrared stereo cameras, an infrared projector, and a color sensor. The infrared stereo cameras generate depth images, and the color sensor generates RGB images, both with a resolution of 1,280 &#x00D7; 720. RGB and depth field images under natural conditions were obtained at wheat tillering and jointing stages. Image collection was carried out under clear and windless weather conditions. The camera was 70 cm above the crop canopy, and set up in the field as shown in <xref ref-type="fig" rid="F2">Figure 2B</xref>. Images were transmitted to a computer in real time via USB 3.0.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p><bold>(A)</bold> Intel RealSense D415; <bold>(B)</bold> Equipment setup for field image acquisition.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g002.tif"/>
</fig>
<p>Since the RGB and depth images had different origins, there is a mismatch problem between the data of different modalities, that is, the same object has a certain degree of position deviation on the images of different modalities. In order to share unified labeling results between RGB images and depth images in subsequent image samples and reduce the impact of mismatch on subsequent feature fusion, coordinate transformation is used first to align depth images and RGB images. The process of image alignment is firstly to restore the depth point of the depth image coordinate system to the world coordinate system, and then to convert the depth point of the world coordinate system to the RGB image coordinate system. The internal parameters matrix, rotation matrix, and translation vector of the depth camera and RGB camera were obtained from the System Design Kit (SDK) provided by Intel. Data loss (void) occurred in depth images due to lighting conditions, infrared reflective properties of object surface materials, and shielding, and depth images were repaired using a hole filling (HF) algorithm (<xref ref-type="bibr" rid="B46">Xu et al., 2020b</xref>) to obtain complete depth information for subsequent feature extraction.</p>
</sec>
<sec id="S2.SS3">
<title>Recoding Depth Images</title>
<p>A depth image contains data captured by a depth camera that reflect the distance between the object and the camera. A depth image provides information on object shape and geometry that are lost in RGB images but crucial for object detection. In many object detection studies based on multi-modal information, depth images provide supplementary information to RGB images and improve the performance of object detection (<xref ref-type="bibr" rid="B16">Gupta et al., 2010</xref>; <xref ref-type="bibr" rid="B21">Hedau et al., 2010</xref>). However, the original depth information is less representative. In particular, feature extraction from depth images with a CNN generates feature maps of distance rather than geometric structures with physical significance. Therefore, single-channel depth images were transformed to three-channel images by recoding the original images to make them more representative and structurally similar to RGB images. The three channels of recoded images are phase, height above ground, and angle with gravity, and recoded images are referred to as PHA images. Phase was calculated according to the mechanism by which depth information is generated (<xref ref-type="bibr" rid="B6">Cai et al., 2017</xref>),</p>
<disp-formula id="S2.E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x22C5;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x03C0;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03D5;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>n</italic> &#x2208; <italic>&#x2115;</italic>, <italic>l</italic> &#x2208; <italic>&#x2115;</italic>, and 2&#x03C0;<italic>l</italic> is the uniqueness range of the camera. The natural number <italic>n</italic> ensures &#x03D5; &#x2208; [0,2&#x03C0;]. The maximum distance in the depth image <italic>d</italic><sub><italic>max</italic></sub> was identified, and the relative distance between object and sensor was converted to height above ground,</p>
<disp-formula id="S2.E2">
<label>(2)</label>
<mml:math id="M2">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>max</mml:mi>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>d</italic>(<italic>i</italic>,<italic>j</italic>) is the depth in image coordinates (<italic>i</italic>,<italic>j</italic>). The third channel (angle with gravity) is the angle between the local surface of a pixel and the direction of gravity (<xref ref-type="bibr" rid="B17">Gupta et al., 2013</xref>),</p>
<disp-formula id="S2.E3">
<label>(3)</label>
<mml:math id="M3">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mo movablelimits="false">min</mml:mo>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>:</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mo>||</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo>||</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:munder>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:msup>
<mml:mi>cos</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03B8;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:msup>
<mml:mi>sin</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03B8;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>g</italic> is the direction of gravity, <italic>N</italic><sub><italic>1</italic></sub> represents the set of normals parallel to the direction of gravity, <italic>N</italic><sub><italic>2</italic></sub> represents the set of normals perpendicular to the direction of gravity, <italic>n</italic><sub><italic>1</italic></sub> and <italic>n</italic><sub><italic>2</italic></sub> represent some element in <italic>N</italic><sub><italic>1</italic></sub> and <italic>N</italic><sub><italic>2</italic></sub>, respectively. And &#x03B8; is the Angle between two vectors.</p>
</sec>
<sec id="S2.SS4">
<title>Weeds Object Detection Network Based on Multi-Modal Information</title>
<p>There are two primary approaches to detect objects based on multi-modal information from RGB-D: (1) to use a depth image as an additional channel of the RGB image (<xref ref-type="bibr" rid="B9">Couprie et al., 2014</xref>); and (2) to separately learn features from RGB and depth images (<xref ref-type="bibr" rid="B44">Wang et al., 2015</xref>). However, these methods can neither extract fine geometric features from depth images nor make full use of the complementarity of different modalities. We designed a network by considering the common features (complementarity) between the two modalities (RGB and depth) and the unique features (independence) learned from single modalities.</p>
<p>The network architecture is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. We designed a three-channel CNN to learn different features from multi-modal information, including two channels for learning RGB- and depth-specific features, and one for learning RGB-D-correlated features. Each network was designed based on Faster R-CNN with VGG16 as a backbone. Since the object area of weeds varied greatly and some were thin and small, we introduced multiscale object detection when designing the correlated detection net. The concept of multiscale representation in CNNs has been demonstrated in previous studies. Low-level feature maps have smaller receptive fields and larger scales, and they contain less semantic information and more low-level feature information such as edges and colors. High-level feature maps contain more semantic information and high-level feature information such as object parts and components (<xref ref-type="bibr" rid="B5">Bell et al., 2016</xref>; <xref ref-type="bibr" rid="B26">Kong et al., 2016</xref>). Therefore, object detection using the last feature map is not favorable for the detection of multiscale and small objects. Therefore, when designing the architecture of correlated detection net, the basic idea is to utilize the advantages of different receptive fields in detecting targets of different scales so that the network can better deal with multi-scale targets and improve the overall detection accuracy. We used the structure of a hypernet as a reference in designing the structure of a correlated detection net (<xref ref-type="bibr" rid="B26">Kong et al., 2016</xref>), fusing the feature maps after the first, third, and fifth convolutional blocks in RGB- and depth-specific detection nets. Because layers had different feature map dimensions, max pool was used in the first layer, and deconv in the fifth layer, to facilitate calculation. To enhance the learning of complementary features by the correlated detection net, instead of directly connecting via add and concat, a previously described method (<xref ref-type="bibr" rid="B47">Xu et al., 2017</xref>) was used in the ultimate fusion of feature maps, and the fused feature map was defined as:</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Weeds detection network architecture.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g003.tif"/>
</fig>
<disp-formula id="S2.E4">
<label>(4)</label>
<mml:math id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2218;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>f</italic><sub><italic>RGB</italic></sub> and <italic>f</italic><sub><italic>Depth</italic></sub> denote the feature maps generated by CNN in RGB- and depth-specific detection nets, respectively, and &#x00B0; denotes the Hadamard product.</p>
<p>In order to generate multi-modal object proposals, three Region Proposal Networks (RPNs; <xref ref-type="bibr" rid="B35">Ren et al., 2017</xref>) are slid over last feature maps. One is for modality-correlated object estimation and the other two are for modality-specific object estimation. The loss function of RPN network is divided into two parts: the boundary-box regression loss function and the classification loss function. For bounding box regression, given an anchor box with (<italic>x</italic><sub><italic>a</italic></sub>,<italic>y</italic><sub><italic>a</italic></sub>,<italic>w</italic><sub><italic>a</italic></sub>,<italic>h</italic><sub><italic>a</italic></sub>), bounding box regression is developed to predict deviations <inline-formula><mml:math id="INEQ11"><mml:mrow><mml:msup><mml:mi>t</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>t</mml:mi><mml:mi>x</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>t</mml:mi><mml:mi>y</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>t</mml:mi><mml:mi>w</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> following (<xref ref-type="bibr" rid="B11">Felzenszwalb et al., 2010</xref>; <xref ref-type="bibr" rid="B14">Girshick et al., 2013</xref>):</p>
<disp-formula id="S2.E5">
<label>(5)</label>
<mml:math id="M5">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable displaystyle="true" rowspacing="0pt">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msubsup>
<mml:mi>t</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msup>
<mml:mo>-</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msup>
<mml:mo>-</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msubsup>
<mml:mi>t</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>w</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msubsup>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mi/>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>x</italic>,<italic>y</italic>,<italic>w</italic> and <italic>h</italic> denote the bounding box&#x2019;s center coordinates and its width and height. <italic>x</italic><sup>&#x2217;</sup>,<italic>y</italic><sup>&#x2217;</sup>,<italic>w</italic><sup>&#x2217;</sup>,<italic>h</italic><sup>&#x2217;</sup> and <italic>x</italic><sub><italic>a</italic></sub>,<italic>y</italic><sub><italic>a</italic></sub>,<italic>w</italic><sub><italic>a</italic></sub>,<italic>h</italic><sub><italic>a</italic></sub> are for the ground-truth box and anchor box, respectively. <italic>Smooth L</italic>1 (<xref ref-type="bibr" rid="B13">Girshick, 2015</xref>) is adopted to calculate the bounding box regression loss.</p>
<disp-formula id="S2.E6">
<label>(6)</label>
<mml:math id="M6">
<mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mtext></mml:mtext>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable displaystyle="true" rowspacing="0pt">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo separator="true">&#x2003;&#x2003;&#x2002;&#x2005;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&lt;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>-</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
</mml:mrow>
<mml:mo mathvariant="italic" separator="true">&#x2003;&#x2003;</mml:mo>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mi/>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>With these definitions, the object estimation multi-task loss <italic>L</italic> is defined as:</p>
<disp-formula id="S2.Ex1">
<label>(7)</label>
<mml:math id="M7">
<mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mfrac>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03BB;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mfrac>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:mrow>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mtext></mml:mtext>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:msubsup>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x002A;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where the mini-batch size is ignored. <italic>i</italic> represents the index of an anchor point, <italic>p</italic><sub><italic>i</italic></sub> and <inline-formula><mml:math id="INEQ16"><mml:msubsup><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup></mml:math></inline-formula> are the predicted object probability of an anchor and ground-truth label. If the anchor is positive, <inline-formula><mml:math id="INEQ17"><mml:mrow><mml:msubsup><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>. And if the anchor is negative, <inline-formula><mml:math id="INEQ18"><mml:mrow><mml:msubsup><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x002A;</mml:mo></mml:msubsup><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>. Two types of anchors are treated as positive: the anchors with the highest intersection over union (<italic>IoU</italic>) overlap with a ground-truth box, and the ones that have an <italic>IoU</italic> overlap higher than 0.7 with any ground-truth box. <italic>L</italic><sub><italic>cls</italic></sub> is log loss over. The two terms in Eq. (7) are normalized by <italic>N</italic><sub><italic>cls</italic></sub> and <italic>N</italic><sub><italic>reg</italic></sub> and weighted by a balancing parameter &#x03BB;. The former is normalized by the mini-batch size and the latter is normalized by the number of anchor locations. The modality-correlated RPN and modality-specific RPNs are trained simultaneously with the same supervision.</p>
<p>To better exploit the complementarity of multi-modal data, we fused data at the decision level of the algorithm for ensemble learning and assigned weights to detection results of the three-channel network. The equation is as follows:</p>
<disp-formula id="S2.E8">
<label>(8)</label>
<mml:math id="M8">
<mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03B1;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x03B2;</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>-</mml:mo>
<mml:mi mathvariant="normal">&#x03B1;</mml:mi>
<mml:mo>-</mml:mo>
<mml:mi mathvariant="normal">&#x03B2;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <italic>g</italic>(&#x22C5;) is the output of detection network, &#x03B1;, &#x03B2; (&#x03B1;&#x2265;0,&#x03B2;&#x2265;0,&#x03B1; + &#x03B2;&#x2264;1) are the ensemble weights for RGB and depth branch, respectively. &#x03B1;, &#x03B2; vary in [0,1] with a step size of 0.05.</p>
</sec>
<sec id="S2.SS5">
<title>Datasets and Model Training Methods</title>
<p>To increase the robustness of the object detection network, we performed image data enhancement by rotation and flipping. The resulting multi-modal weeds in the wheat field dataset (MWWFD) included 1,228 RGB images (500 &#x00D7; 500) and 1,228 corresponding depth images (500 &#x00D7; 500). Broad-leaf and grass weeds in images were labeled using LabelImg; 1,105 images were used for training, and 123 images for testing.</p>
<p>The deep learning framework used is TensorFlow, GPU is NVIDIA RTX2080Ti, CPU is Intel(R) Core(TM) i7-7800x CPU @ 3.50 GHz. The multi-modal weeds detection network was subjected to end-to-end training with backpropagation and stochastic gradient descent methods. For RPN networks, each mini-batch arises from a single image that contains many positive and negative example anchors. Some proposals generated by the region proposal network (RPN) overlapped significantly. To reduce redundancy, we performed non-maximum suppression (NMS) and set the threshold of <italic>IoU</italic> at 0.7. Other training hyperparameters are shown in <xref ref-type="table" rid="T1">Table 1</xref>. We adopted the weight sharing strategy in the training process, which has been proven effective in many studies because it greatly reduces model complexity and running time (<xref ref-type="bibr" rid="B27">Lecun and Bottou, 1998</xref>). <xref ref-type="bibr" rid="B18">Gupta et al. (2016)</xref> proved that features learned from depth images are complementary to RGB features even if a CNN based on depth images is supervised and trained by a CNN based on RGB images, and training a network with shared weights is effective.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Training parameters of multi-modal weeds detection network.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Parameters</td>
<td valign="top" align="left">Value</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Initial learning rate</td>
<td valign="top" align="left">0.001</td>
</tr>
<tr>
<td valign="top" align="left">Momentum</td>
<td valign="top" align="left">0.9</td>
</tr>
<tr>
<td valign="top" align="left">Weight decay</td>
<td valign="top" align="left">0.0001</td>
</tr>
<tr>
<td valign="top" align="left">Iteration per epoch</td>
<td valign="top" align="left">1000</td>
</tr>
<tr>
<td valign="top" align="left">Number of epochs</td>
<td valign="top" align="left">300</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S2.SS6">
<title>Evaluation Methods</title>
<p>Mean average precision (<italic>mAP</italic>) is a common and reliable measure of model performance in the detection of multi-category objects.</p>
<disp-formula id="S2.E9">
<label>(9)</label>
<mml:math id="M9">
<mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mo largeop="true" symmetric="true">&#x222B;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo mathvariant="italic" rspace="0pt">d</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>P</italic> denotes the precision and <italic>R</italic> denotes the recall. <italic>m</italic>AP is APs averaged over all categories. However, in the detection of weeds in wheat fields, labeling was complicated by the cluster growth of weeds. As shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, labeling affected the evaluation of detection precision. Therefore, we used <italic>mAP</italic> to evaluate model detection performance, and intersection over ground truth (<italic>IoG</italic>), which is the quotient of the intersection and union of the detected and labeled datasets, to evaluate the overall precision of weed detection.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p><bold>(A,B)</bold> Different labeling results of the same image.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g004.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="results" id="S3">
<title>Results</title>
<sec id="S3.SS1">
<title>Evaluation of PHA Image Quality</title>
<p>We compared the structures of PHA and RGB images from two aspects for suitability in CNN-based feature learning. As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, the entropy values of PHA and RGB images were closer to each other than to that of depth images, suggesting that they contained more information and were more closely correlated than depth images. Comparison of output from the first convolutional layer of VGG16 showed that PHA images well retained the height information in depth images, with yellow areas in the feature map representing a higher wheat canopy (<xref ref-type="fig" rid="F6">Figure 6</xref>). The depth images had pixels with uniform color in soil and weeds areas, while PHA and RGB images had similar textures, which also indicated their similarity. These comparisons indicated that PHA images obtained by recoding depth images were similar to RGB images in terms of information and structure and were more suitable than depth images for CNN-based feature learning.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Entropy values of PHA, RGB, and depth images.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Feature maps from first convolutional layers in VGG16 network of PHA, RGB, and depth images.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g006.tif"/>
</fig>
</sec>
<sec id="S3.SS2">
<title>Detection of Weeds in Wheat Fields With Different Datasets</title>
<p>The precision of weeds detection with different datasets is shown in <xref ref-type="table" rid="T2">Table 2</xref>. Detection based on the PHA dataset was significantly better than on the depth dataset, which confirmed that PHA was more suitable for CNN-based feature learning. Comparison of weeds detection with the three single-modal datasets showed that RGB images were superior in the detection of broad-leaf weeds. Depth and PHA images had similar detection performance regardless of weeds species, and PHA images had the best results in grass weeds detection. The geometric features extracted from PHA images effectively distinguished wheat from the various weeds species.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Weeds detection with single-modal datasets.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Dataset</td>
<td valign="top" align="center">Backbone</td>
<td valign="top" align="center">mAP of broad-leaf weeds (%)</td>
<td valign="top" align="center">mAP of grass weeds (%)</td>
<td valign="top" align="center">IoG (%)</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">RGB</td>
<td valign="top" align="center">VGG16</td>
<td valign="top" align="center">38.5</td>
<td valign="top" align="center">24.7</td>
<td valign="top" align="center">77.6</td>
</tr>
<tr>
<td valign="top" align="left">Depth</td>
<td valign="top" align="center">VGG16</td>
<td valign="top" align="center">11.6</td>
<td valign="top" align="center">11.7</td>
<td valign="top" align="center">42.1</td>
</tr>
<tr>
<td valign="top" align="left">PHA</td>
<td valign="top" align="center">VGG16</td>
<td valign="top" align="center">24.6</td>
<td valign="top" align="center">25.2</td>
<td valign="top" align="center">56.9</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S3.SS3">
<title>Detection With Multi-Modal Datasets in Multichannel Network Architectures</title>
<p>We compared the detection of weeds in wheat fields with different multi-modal datasets and network architectures (<xref ref-type="table" rid="T3">Table 3</xref>). Dual-VGG16 is the direct superimposition of the last layers of feature maps of different modal images regardless of feature learning in the remaining convolutional layers. Direct superimposition of feature maps reduced precision compared to detection based on single-modal RGB images (<xref ref-type="table" rid="T2">Table 2</xref>). This is consistent with previous work (<xref ref-type="bibr" rid="B18">Gupta et al., 2016</xref>) showing that information mapping in the same scene varies across modalities, and direct fusion can cause the divergence of detection results and diminished precision. Therefore, the complementarity of different modality datasets was considered in network design, and detection performance was optimized by fusing feature maps from different convolutional layers. By comparing the performance of the same model in different datasets, it can be proved that the PHA image obtained by recoding is more conducive to weeds detection. The detection precision (mAP) of grass and broad-leaf weeds with correlated detection net (RPN-Corr) was 29.9 and 39.3%, respectively, and the overall precision (IoG) was 81.4%.</p>
<table-wrap position="float" id="T3">
<label>TABLE 3</label>
<caption><p>Weed detection with multi-modal datasets in multichannel network.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Dataset</td>
<td valign="top" align="center">Backbone</td>
<td valign="top" align="center">mAP of broad-leaf weeds (%)</td>
<td valign="top" align="center">mAP of grass weeds (%)</td>
<td valign="top" align="center">IoG (%)</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">RGB-D</td>
<td valign="top" align="center">Dual-VGG16</td>
<td valign="top" align="center">36.2</td>
<td valign="top" align="center">23.5</td>
<td valign="top" align="center">72.6</td>
</tr>
<tr>
<td valign="top" align="left">RGB-PHA</td>
<td valign="top" align="center">Dual-VGG16</td>
<td valign="top" align="center">37.1</td>
<td valign="top" align="center">24.2</td>
<td valign="top" align="center">73.5</td>
</tr>
<tr>
<td valign="top" align="left">RGB-D</td>
<td valign="top" align="center">RPN-Corr</td>
<td valign="top" align="center">37.9</td>
<td valign="top" align="center">25.1</td>
<td valign="top" align="center">75.2</td>
</tr>
<tr>
<td valign="top" align="left">RGB-PHA</td>
<td valign="top" align="center">RPN-Corr</td>
<td valign="top" align="center">39.3</td>
<td valign="top" align="center">29.9</td>
<td valign="top" align="center">81.4</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S3.SS4">
<title>Ensemble Learning Strategy</title>
<p>While taking into account the complementarity of multi-modal datasets, the independence of datasets was exploited through an ensemble learning strategy. Three independent detection models (RGB-specific, depth-specific, and RGB-D-correlated) were trained, and weights were assigned according to Eq. (3), with results as shown in <xref ref-type="fig" rid="F7">Figure 7</xref>. The detection precision was improved compared to a correlated detection net. When &#x03B1; = 0.4 and &#x03B2; = 0.3, <italic>mAP</italic> = 39.6%. The detection precision of broad-leaf and grass weeds was 42.9% and 36.1%, respectively, and <italic>IoG</italic> = 89.3%. <xref ref-type="fig" rid="F8">Figure 8</xref> shows weeds results in different images. Notably, we still found some false positive cases on the test set. These cases may be caused by labeling errors. Due to complex field conditions and low image resolution, some fine weeds may be omitted in labeling. The existence of multiple labeling methods in the same weed area is also the main reason for false positive cases. Therefore, we proposed a new accuracy evaluation method, hoping to avoid the influence of such situation on detection results. In the second row of <xref ref-type="fig" rid="F8">Figure 8</xref>, we can find that our method can well realize weed detection in wheat field under natural environment when there is no labeling information. Most areas of both grass and broad-leaf weeds were detected, and the majority of wheat leaves were correctly recognized even with the overlap of leaves in the fields.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Ensemble learning results of RGB-specific, depth-specific, and RGB-D-correlated models with different weight assignments.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g007.tif"/>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption><p>Weeds detection with images of wheat fields. The three graphs in the first row show the detection results of the test set, where green indicates true positive cases, blue indicates corresponding labeling results, pink indicates false negative cases, and red indicates false positive cases. The two graphs in the second row show the detection results of our method on images outside the MWWFD.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-732968-g008.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="S4">
<title>Discussion</title>
<sec id="S4.SS1">
<title>Weeds Detection in Wheat Fields Based on Multi-Modal Information</title>
<p>Previous work in the accurate detection of weeds in wheat fields with information technology mostly used single modalities, such as spectral information and RGB images. However, because of the similar leaf shape and canopy structure of grass weeds and wheat, there are few differences in RGB image features and in reflectance spectra at characteristic wavelengths, which makes the use of modal information difficult for grass weeds detection (<xref ref-type="bibr" rid="B15">G&#x00F3;mez-Casero et al., 2010</xref>). We fused depth images with RGB images, extracted geometric features such as height from PHA images, and used multi-modal information for the effective detection of weeds in wheat fields. In the proposed three-channel weeds detection network, feature maps from different convolutional layers were fused using the concept of multiscale object detection. Ensemble learning was carried out at the decision level based on the independence and complementarity of different modalities, which effectively improved weeds detection precision. However, weight assignment in ensemble learning still relied on hand-designed weight gradient experiments, and detection precision remained suboptimal. Weight assignment methods should be further studied.</p>
</sec>
<sec id="S4.SS2">
<title>Application of Different Machine Learning Algorithms in Weeds Detection</title>
<p>The selection and improvement of machine learning algorithms are a focus in the development of weeds detection technologies. The accuracy and real-time of the detection algorithm determine whether it can be applied in practical agricultural production. In recent years, deep learning methods based on CNN have been widely used in weeds detection with the advantage of end-to-end, avoiding the influence of extraction of manually designed features on detection results (<xref ref-type="bibr" rid="B22">Huang et al., 2018a</xref>). This involves calculation of coordinates of bounding boxes around object objects and generates detection results in which the size of the predicted bounding box matches that of the actual weeds object, which improves the precision of weeds detection (<xref ref-type="bibr" rid="B19">Hall et al., 2017</xref>). We added a input channel and multi-modal feature fusion blocks, and realized high-precision weeds detection in wheat field through use of multi-modal information effectively. However, compared to traditional machine learning algorithms (<xref ref-type="bibr" rid="B38">Siddiqi et al., 2014</xref>; <xref ref-type="bibr" rid="B45">Xu et al., 2020a</xref>), the proposed weeds detection algorithm still suffers from a high computational load and low computational efficiency. Although weight sharing was used in training to reduce the computational burden, the demand on the hardware was still high. In future work, we will explore model compression to improve detection efficiency while maintaining precision.</p>
</sec>
<sec id="S4.SS3">
<title>Weeds Detection in Complex Wheat Field Background</title>
<p>Data (images) of a single category and simple background are usually used in weeds detection, which is limited to the early stage of wheat growth (<xref ref-type="bibr" rid="B29">Nieuwenhuizen et al., 2010</xref>; <xref ref-type="bibr" rid="B42">Tellaeche et al., 2011</xref>). We considered two growth periods with high incidence of weeds in wheat fields, and the cultivation conditions of field experiments were in line with the actual situation. A weeds detection model based on multiscale object detection was suitable to detect weeds areas of different sizes. However, due to shielding of wheat and weeds leaves in fields, RGB-D images from a single perspective failed to capture the information of the shielded objects. Therefore, we will adopt a multi-perspective approach for image acquisition to mitigate the effect of leaf shielding on weeds detection.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="S5">
<title>Conclusion</title>
<p>We proposed a three-channel weeds detection method based on multi-modal information by fusing RGB and depth images and applying the concept of multiscale object detection, which effectively improved the precision of weeds detection in wheat fields. The single-channel depth image is recoded, and the resulting PHA images were more similar in structure to RGB images and more suitable for CNN-based feature learning. The results showed that when the same network was used, weeds detection precision based on PHA images was 1.35-fold of that based on depth images. And the independence and complementarity of the two modalities of RGB and depth images were taken into account, and a three-channel weeds detection network was designed from the perspective of feature- and decision-level fusion. The results showed that the model could effectively detect different species of weeds in wheat fields (<italic>IoG</italic> = 89.3%).</p>
</sec>
<sec sec-type="data-availability" id="S6">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="S7">
<title>Author Contributions</title>
<p>JN, YZ, XJ, and WC designed the research. JN and KX developed the algorithms. KX, ZJ, and SL performed the research. KX analyzed the data and wrote the manuscript. All authors have read and approved the final manuscript.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="S13">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="S12">
<title>Funding</title>
<p>This work was supported in part by the National Key Research and Development Program of China (Grant No. 2017YFD0201501), National Natural Science Foundation of China (Grant No. 31871524), Six Talent Peaks Project in Jiangsu Province (Grant No. XYDXX-049), Primary Research and Development Plan of Jiangsu Province of China (BE2017385, BE2018399, and BE2019306), and the 111 Project (B16026).</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alessandro</surname> <given-names>D.</given-names></name> <name><surname>Freitas</surname> <given-names>D. M.</given-names></name> <name><surname>Gercina</surname> <given-names>G.</given-names></name> <name><surname>Pistori</surname> <given-names>H.</given-names></name> <name><surname>Folhes</surname> <given-names>M. T.</given-names></name></person-group> (<year>2017</year>). <article-title>Weed detection in soybean crops using ConvNets.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>143</volume> <fpage>314</fpage>&#x2013;<lpage>324</lpage>. <pub-id pub-id-type="doi">10.1002/ps.3839</pub-id> <pub-id pub-id-type="pmid">24889377</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alsamhi</surname> <given-names>S. H.</given-names></name> <name><surname>Almalki</surname> <given-names>F. A.</given-names></name> <name><surname>Al-Dois</surname> <given-names>H.</given-names></name> <name><surname>Ben Othman</surname> <given-names>S.</given-names></name> <name><surname>Hassan</surname> <given-names>J.</given-names></name> <name><surname>Hawbani</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Machine Learning for Smart Environments in B5G Networks: connectivity and QoS.</article-title> <source><italic>Comput. Intell. Neurosci.</italic></source> <volume>2021</volume>:<issue>6805151</issue>. <pub-id pub-id-type="doi">10.1155/2021/6805151</pub-id> <pub-id pub-id-type="pmid">34589123</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bah</surname> <given-names>M. D.</given-names></name> <name><surname>Hafiane</surname> <given-names>A.</given-names></name> <name><surname>Canals</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep Learning with Unsupervised Data Labeling for Weed Detection in Line Crops in UAV Images.</article-title> <source><italic>Remote Sens.</italic></source> <volume>10</volume>:<issue>1690</issue>. <pub-id pub-id-type="doi">10.3390/rs10111690</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bakhshipour</surname> <given-names>A.</given-names></name> <name><surname>Jafari</surname> <given-names>A.</given-names></name> <name><surname>Nassiri</surname> <given-names>S. M.</given-names></name> <name><surname>Zare</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Weed segmentation using texture features extracted from wavelet sub-images.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>157</volume> <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2017.02.002</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bell</surname> <given-names>S.</given-names></name> <name><surname>Zitnick</surname> <given-names>C. L.</given-names></name> <name><surname>Bala</surname> <given-names>K.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Inside-Outside Net: Detecting Objects in Context with Skip Pooling and Recurrent Neural Networks</article-title>,&#x201D; in <source><italic>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, (<publisher-loc>Piscataway</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cai</surname> <given-names>Z.</given-names></name> <name><surname>Cai</surname> <given-names>Z.</given-names></name> <name><surname>Shao</surname> <given-names>L.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>RGB-D data fusion in complex space</article-title>,&#x201D; in <source><italic>IEEE International Conference on Image Processing, 2017</italic></source>, (<publisher-loc>Beijing</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Camille</surname> <given-names>L.</given-names></name> <name><surname>Philippe</surname> <given-names>B.</given-names></name> <name><surname>Guillaume</surname> <given-names>J.</given-names></name> <name><surname>Bruno</surname> <given-names>R.</given-names></name> <name><surname>Sylvain</surname> <given-names>L.</given-names></name> <name><surname>Fr&#x00E9;d&#x00E9;ric</surname> <given-names>B.</given-names></name></person-group> (<year>2008</year>). <article-title>Assessment of Unmanned Aerial Vehicles Imagery for Quantitative Monitoring of Wheat Crop in Small Plots.</article-title> <source><italic>Sensors</italic></source> <volume>8</volume> <fpage>3557</fpage>&#x2013;<lpage>3585</lpage>. <pub-id pub-id-type="doi">10.3390/s8053557</pub-id> <pub-id pub-id-type="pmid">27879893</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christensen</surname> <given-names>S.</given-names></name> <name><surname>S&#x00F8;gaard</surname> <given-names>H.</given-names></name> <name><surname>Kudsk</surname> <given-names>P.</given-names></name> <name><surname>N&#x00F8;rremark</surname> <given-names>M.</given-names></name> <name><surname>Lund</surname> <given-names>I.</given-names></name> <name><surname>Nadimi</surname> <given-names>E. S.</given-names></name><etal/></person-group> (<year>2010</year>). <article-title>Site-specific weed control technologies.</article-title> <source><italic>Weed Res.</italic></source> <volume>49</volume> <fpage>233</fpage>&#x2013;<lpage>241</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-3180.2009.00696.x</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Couprie</surname> <given-names>C.</given-names></name> <name><surname>Najman</surname> <given-names>L.</given-names></name> <name><surname>Lecun</surname> <given-names>Y.</given-names></name></person-group> (<year>2014</year>). <article-title>Convolutional Nets and Watershed Cuts for Real-Time Semantic Labeling of RGBD Videos.</article-title> <source><italic>J. Mach. Learn. Res.</italic></source> <volume>15</volume> <fpage>3489</fpage>&#x2013;<lpage>3511</lpage>.</citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fahad</surname> <given-names>S.</given-names></name> <name><surname>Hussain</surname> <given-names>S.</given-names></name> <name><surname>Chauhan</surname> <given-names>B. S.</given-names></name> <name><surname>Saud</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Hassan</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Weed growth and crop yield loss in wheat as influenced by row spacing and weed emergence times.</article-title> <source><italic>Crop Protect.</italic></source> <volume>71</volume> <fpage>101</fpage>&#x2013;<lpage>108</lpage>. <pub-id pub-id-type="doi">10.1016/j.cropro.2015.02.005</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Felzenszwalb</surname> <given-names>P. F.</given-names></name> <name><surname>Girshick</surname> <given-names>R. B.</given-names></name> <name><surname>McAllester</surname> <given-names>D.</given-names></name> <name><surname>Ramanan</surname> <given-names>D.</given-names></name></person-group> (<year>2010</year>). <article-title>Object Detection with Discriminatively Trained Part-Based Models.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>32</volume> <fpage>1627</fpage>&#x2013;<lpage>1645</lpage>.</citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gaba</surname> <given-names>S.</given-names></name> <name><surname>Chauvel</surname> <given-names>B.</given-names></name> <name><surname>Dessaint</surname> <given-names>F.</given-names></name> <name><surname>Bretagnolle</surname> <given-names>V.</given-names></name> <name><surname>Petit</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Weed species richness in winter wheat increases with landscape heterogeneity.</article-title> <source><italic>Agric. Ecosyst. Environ.</italic></source> <volume>138</volume> <fpage>318</fpage>&#x2013;<lpage>323</lpage>.</citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). &#x201C;<article-title>Fast R-CNN</article-title>,&#x201D; in <source><italic>Computer Science 2015 IEEE International Conference on Computer Vision (ICCV)</italic></source>, (<publisher-loc>Santiago</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Donahue</surname> <given-names>J.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name> <name><surname>Malik</surname> <given-names>J.</given-names></name></person-group> (<year>2013</year>). &#x201C;<article-title>Rich Feature Hierarchies for Accurate Object Detection and Semantic Segmentation</article-title>,&#x201D; in <source><italic>IEEE Conference on Computer Vision and Pattern Recognition</italic></source>, (<publisher-loc>Columbus</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>G&#x00F3;mez-Casero</surname> <given-names>M.</given-names></name> <name><surname>Castillejo-Gonz&#x00E1;lez</surname> <given-names>I.</given-names></name> <name><surname>Garc&#x00ED;a-Ferrer</surname> <given-names>A.</given-names></name> <name><surname>Pe?a-Barrag&#x00E1;n</surname> <given-names>J. M.</given-names></name> <name><surname>Jurado-Exp&#x00F3;sito</surname> <given-names>M.</given-names></name> <name><surname>Garc&#x00ED;a-Torres</surname> <given-names>L.</given-names></name><etal/></person-group> (<year>2010</year>). <article-title>Spectral discrimination of wild oat and canary grass in wheat fields for less herbicide application.</article-title> <source><italic>Agron. Sustain. Dev.</italic></source> <volume>30</volume> <fpage>689</fpage>&#x2013;<lpage>699</lpage>.</citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname> <given-names>A.</given-names></name> <name><surname>Hebert</surname> <given-names>M.</given-names></name> <name><surname>Kanade</surname> <given-names>T.</given-names></name> <name><surname>Blei</surname> <given-names>D. M.</given-names></name></person-group> (<year>2010</year>). &#x201C;<article-title>Estimating spatial layout of rooms using volumetric reasoning about objects and surfaces</article-title>,&#x201D; in <source><italic>Proceedings of (NeurIPS) Neural Information Processing Systems</italic></source>, (<publisher-loc>Pittsburgh</publisher-loc>: <publisher-name>Carnegie Mellon University</publisher-name>), <fpage>1288</fpage>&#x2013;<lpage>1296</lpage>.</citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname> <given-names>S.</given-names></name> <name><surname>Arbelaez</surname> <given-names>P.</given-names></name> <name><surname>Malik</surname> <given-names>J.</given-names></name></person-group> (<year>2013</year>). &#x201C;<article-title>Perceptual Organization and Recognition of Indoor Scenes from RGB-D Images</article-title>,&#x201D; in <source><italic>26th IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, (<publisher-loc>Portland</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>564</fpage>&#x2013;<lpage>571</lpage>.</citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname> <given-names>S.</given-names></name> <name><surname>Hoffman</surname> <given-names>J.</given-names></name> <name><surname>Malik</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Cross Modal Distillation for Supervision Transfer</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</italic></source>, (<publisher-loc>Las Vegas</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2827</fpage>&#x2013;<lpage>2836</lpage>.</citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hall</surname> <given-names>D.</given-names></name> <name><surname>Dayoub</surname> <given-names>F.</given-names></name> <name><surname>Kulk</surname> <given-names>J.</given-names></name> <name><surname>Mccool</surname> <given-names>C.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Towards unsupervised weed scouting for agricultural robotics</article-title>,&#x201D; in <source><italic>2017 IEEE International Conference on Robotics &#x0026; Automation</italic></source>, (<publisher-loc>Piscataway</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haque</surname> <given-names>A.</given-names></name> <name><surname>Milstein</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>F.-F.</given-names></name></person-group> (<year>2020</year>). <article-title>Illuminating the dark spaces of healthcare with ambient intelligence.</article-title> <source><italic>Nature</italic></source> <volume>585</volume> <fpage>193</fpage>&#x2013;<lpage>202</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2669-y</pub-id> <pub-id pub-id-type="pmid">32908264</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hedau</surname> <given-names>V.</given-names></name> <name><surname>Hoiem</surname> <given-names>D.</given-names></name> <name><surname>Forsyth</surname> <given-names>D.</given-names></name></person-group> (<year>2010</year>). &#x201C;<article-title>Thinking inside the box: Using appearance models and context based on room geometry</article-title>,&#x201D; in <source><italic>ECCV&#x2019;10: Proceedings of the 11th European conference on Computer vision: Part VI</italic></source>, (<publisher-loc>Heraklion</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>224</fpage>&#x2013;<lpage>237</lpage>.</citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Lan</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>A.</given-names></name> <name><surname>Deng</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name></person-group> (<year>2018a</year>). <article-title>A fully convolutional network for weed mapping of unmanned aerial vehicle (UAV) imagery.</article-title> <source><italic>PLoS One</italic></source> <volume>13</volume>:<issue>e0196302</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0196302</pub-id> <pub-id pub-id-type="pmid">29698500</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Lan</surname> <given-names>Y.</given-names></name> <name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>A.</given-names></name> <name><surname>Sheng</surname> <given-names>W.</given-names></name></person-group> (<year>2018b</year>). <article-title>A Semantic Labeling Approach for Accurate Weed Mapping of High Resolution UAV Imagery.</article-title> <source><italic>Sensors</italic></source> <volume>18</volume>:<issue>2113</issue>. <pub-id pub-id-type="doi">10.3390/s18072113</pub-id> <pub-id pub-id-type="pmid">29966392</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jaime</surname> <given-names>R.</given-names></name> <name><surname>Ricardo</surname> <given-names>D. C.</given-names></name></person-group> (<year>2017</year>). <article-title>Glyphosate Residues in Groundwater, Drinking Water and Urine of Subsistence Farmers from Intensive Agriculture Localities: a Survey in Hopelch&#x00E9;n, Campeche, Mexico. International.</article-title> <source><italic>J. Environ. Res. Public Health</italic></source> <volume>14</volume>:<issue>595</issue>. <pub-id pub-id-type="doi">10.3390/ijerph14060595</pub-id> <pub-id pub-id-type="pmid">28587206</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kniss</surname> <given-names>A. R.</given-names></name></person-group> (<year>2017</year>). <article-title>Long-term trends in the intensity and relative toxicity of herbicide use.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>8</volume>:<issue>14865</issue>. <pub-id pub-id-type="doi">10.1038/ncomms14865</pub-id> <pub-id pub-id-type="pmid">28393866</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kong</surname> <given-names>T.</given-names></name> <name><surname>Yao</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>F.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>HyperNet: Towards Accurate Region Proposal Generation and Joint Object Detection</article-title>,&#x201D; in <source><italic>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, (<publisher-loc>Las Vegas</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lecun</surname> <given-names>Y.</given-names></name> <name><surname>Bottou</surname> <given-names>L.</given-names></name></person-group> (<year>1998</year>). <article-title>Gradient-based learning applied to document recognition.</article-title> <source><italic>Proc. IEEE</italic></source> <volume>86</volume> <fpage>2278</fpage>&#x2013;<lpage>2324</lpage>. <pub-id pub-id-type="doi">10.1109/5.726791</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Munier-Jolain</surname> <given-names>N. M.</given-names></name> <name><surname>Guyot</surname> <given-names>S.</given-names></name> <name><surname>Colbach</surname> <given-names>N.</given-names></name></person-group> (<year>2013</year>). <article-title>A 3D model for light interception in heterogeneous crop: weed canopies: model structure and evaluation.</article-title> <source><italic>Ecol. Model.</italic></source> <volume>250</volume> <fpage>101</fpage>&#x2013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.1016/j.ecolmodel.2012.10.023</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nieuwenhuizen</surname> <given-names>A. T.</given-names></name> <name><surname>Hofstee</surname> <given-names>J. W.</given-names></name> <name><surname>Henten</surname> <given-names>E.</given-names></name></person-group> (<year>2010</year>). <article-title>Performance evaluation of an automated detection and control system for volunteer potatoes in sugar beet fields.</article-title> <source><italic>Biosyst. Eng.</italic></source> <volume>107</volume> <fpage>46</fpage>&#x2013;<lpage>53</lpage>.</citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Patr&#x00ED;cioa</surname> <given-names>D.</given-names></name> <name><surname>Riederb</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Computer vision and artificial intelligence in precision agriculture for grain crops: a systematic review.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>153</volume> <fpage>69</fpage>&#x2013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2018.08.001</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Petra</surname> <given-names>B.</given-names></name> <name><surname>Tom</surname> <given-names>D.</given-names></name> <name><surname>Grzegorz</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>Analysis of morphology-based features for classification of crop and weeds in precision agriculture.</article-title> <source><italic>IEEE Robot. Autom. Lett.</italic></source> <volume>3</volume> <fpage>2950</fpage>&#x2013;<lpage>2956</lpage>.</citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pflanz</surname> <given-names>M.</given-names></name> <name><surname>Nordmeyer</surname> <given-names>H.</given-names></name> <name><surname>Schirrmann</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Weed Mapping with UAS Imagery and a Bag of Visual Words Based Image Classifier.</article-title> <source><italic>Remote Sens.</italic></source> <volume>10</volume>:<issue>1530</issue>.</citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Piron</surname> <given-names>A.</given-names></name> <name><surname>Leemans</surname> <given-names>V.</given-names></name> <name><surname>Lebeau</surname> <given-names>F.</given-names></name> <name><surname>Destain</surname> <given-names>M. F.</given-names></name></person-group> (<year>2009</year>). <article-title>Improving in-row weed detection in multispectral stereoscopic images.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>69</volume> <fpage>73</fpage>&#x2013;<lpage>79</lpage>.</citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>C. R.</given-names></name> <name><surname>Wei</surname> <given-names>L.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Hao</surname> <given-names>S.</given-names></name> <name><surname>Guibas</surname> <given-names>L. J.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Frustum PointNets for 3D Object Detection from RGB-D Data</article-title>,&#x201D; in <source><italic>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic></source>, (<publisher-loc>Salt Lake City</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Faster R-CNN: towards Real-Time Object Detection with Region Proposal Networks.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>39</volume> <fpage>1137</fpage>&#x2013;<lpage>1149</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2016.2577031</pub-id> <pub-id pub-id-type="pmid">27295650</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rose</surname> <given-names>M. T.</given-names></name> <name><surname>Cavagnaro</surname> <given-names>T. R.</given-names></name> <name><surname>Scanlan</surname> <given-names>C. A.</given-names></name> <name><surname>Rose</surname> <given-names>T. J.</given-names></name> <name><surname>Vancov</surname> <given-names>T.</given-names></name> <name><surname>Kimber</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Impact of herbicides on soil biology and function.</article-title> <source><italic>Adv. Agron.</italic></source> <volume>136</volume> <fpage>133</fpage>&#x2013;<lpage>220</lpage>.</citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saleh</surname> <given-names>H.</given-names></name> <name><surname>Alharbi</surname> <given-names>A.</given-names></name> <name><surname>Alsamhi</surname> <given-names>S. H.</given-names></name></person-group> (<year>2021</year>). <article-title>OPCNN-FAKE: optimized Convolutional Neural Network for Fake News Detection.</article-title> <source><italic>IEEE Access</italic></source> <volume>9</volume> <fpage>129471</fpage>&#x2013;<lpage>129489</lpage>.</citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Siddiqi</surname> <given-names>M. H.</given-names></name> <name><surname>Lee</surname> <given-names>S. W.</given-names></name> <name><surname>Khan</surname> <given-names>A. M.</given-names></name></person-group> (<year>2014</year>). <article-title>Weed Image Classification using Wavelet Transform, Stepwise Linear Discriminant Analysis, and Support Vector Machines for an Automatic Spray Control System.</article-title> <source><italic>J. Inform. Sci. Eng.</italic></source> <volume>30</volume> <fpage>1227</fpage>&#x2013;<lpage>1244</lpage>.</citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Siddiqui</surname> <given-names>I.</given-names></name> <name><surname>Bajwa</surname> <given-names>R.</given-names></name> <name><surname>Javaid</surname> <given-names>A.</given-names></name></person-group> (<year>2010</year>). <article-title>Effect of six problematic weeds on growth and yield of wheat.</article-title> <source><italic>Pak. J. Bot.</italic></source> <volume>42</volume> <fpage>2461</fpage>&#x2013;<lpage>2471</lpage>.</citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>D. M.</given-names></name> <name><surname>Eade</surname> <given-names>R.</given-names></name> <name><surname>Scaife</surname> <given-names>A. A.</given-names></name> <name><surname>Caron</surname> <given-names>L. P.</given-names></name> <name><surname>Danabasoglu</surname> <given-names>G.</given-names></name> <name><surname>Delsole</surname> <given-names>T. M.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>A comprehensive review on automation in agriculture using artificial intelligence.</article-title> <source><italic>NPJ Clim. Atmos. Sci.</italic></source> <volume>2</volume> <fpage>1</fpage>&#x2013;<lpage>12</lpage>.</citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tellaeche</surname> <given-names>A.</given-names></name> <name><surname>Burgosartizzu</surname> <given-names>X. P.</given-names></name> <name><surname>Pajares</surname> <given-names>G.</given-names></name> <name><surname>Ribeiro</surname> <given-names>A.</given-names></name> <name><surname>Fern&#x00E1;ndez-Quintanilla</surname> <given-names>C.</given-names></name></person-group> (<year>2008</year>). <article-title>A new vision-based approach to differential spraying in precision agriculture.</article-title> <source><italic>Comput. Electron. Agricu.</italic></source> <volume>60</volume> <fpage>144</fpage>&#x2013;<lpage>155</lpage>.</citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tellaeche</surname> <given-names>A.</given-names></name> <name><surname>Pajares</surname> <given-names>G.</given-names></name> <name><surname>Burgos-Artizzu</surname> <given-names>X. P.</given-names></name> <name><surname>Ribeiro</surname> <given-names>A.</given-names></name></person-group> (<year>2011</year>). <article-title>A computer vision approach for weeds identification through Support Vector Machines.</article-title> <source><italic>Appl. Soft Comput.</italic></source> <volume>11</volume> <fpage>908</fpage>&#x2013;<lpage>915</lpage></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ulber</surname> <given-names>L.</given-names></name> <name><surname>Steinmann</surname> <given-names>H. H.</given-names></name> <name><surname>Klimek</surname> <given-names>S.</given-names></name> <name><surname>Isselstein</surname> <given-names>J.</given-names></name></person-group> (<year>2009</year>). <article-title>An on-farm approach to investigate the impact of diversified crop rotations on weed species richness and composition in winter wheat.</article-title> <source><italic>Weed Res.</italic></source> <volume>49</volume> <fpage>534</fpage>&#x2013;<lpage>543</lpage>.</citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>A.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Cai</surname> <given-names>J.</given-names></name> <name><surname>Cham</surname> <given-names>T. J.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Large-Margin Multi-Modal Deep Learning for RGB-D Object Recognition.</article-title> <source><italic>IEEE Trans. Multimedia</italic></source> <volume>17</volume> <fpage>1887</fpage>&#x2013;<lpage>1898</lpage>. <pub-id pub-id-type="doi">10.1109/tmm.2015.2476655</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>K.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Cao</surname> <given-names>W.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>R.</given-names></name> <name><surname>Ni</surname> <given-names>J.</given-names></name></person-group> (<year>2020a</year>). <article-title>Recognition of Weeds in Wheat Fields Based on the Fusion of RGB Images and Depth Images.</article-title> <source><italic>IEEE Access</italic></source> <volume>8</volume> <fpage>110362</fpage>&#x2013;<lpage>110370</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.3001999</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Cao</surname> <given-names>W.</given-names></name> <name><surname>Ni</surname> <given-names>J.</given-names></name></person-group> (<year>2020b</year>). <article-title>Spectrum-and RGB-D-Based Image Fusion for the Prediction of Nitrogen Accumulation in Wheat.</article-title> <source><italic>Remote Sens.</italic></source> <volume>12</volume>:<issue>4040</issue>. <pub-id pub-id-type="doi">10.3390/rs12244040</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>G.</given-names></name> <name><surname>Luo</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Multi-modal deep feature learning for RGB-D object detection.</article-title> <source><italic>Pattern Recognit.</italic></source> <volume>72</volume> <fpage>300</fpage>&#x2013;<lpage>313</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2019.2891104</pub-id> <pub-id pub-id-type="pmid">30624216</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Grift</surname> <given-names>T. E.</given-names></name></person-group> (<year>2012</year>). <article-title>A LIDAR-based crop height measurement system for Miscanthus giganteus.</article-title> <source><italic>Comput. Electron. Agric.</italic></source> <volume>85</volume> <fpage>70</fpage>&#x2013;<lpage>76</lpage>.</citation></ref>
</ref-list>
</back>
</article>