<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Earth Sci.</journal-id>
<journal-title>Frontiers in Earth Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Earth Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-6463</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1363571</article-id>
<article-id pub-id-type="doi">10.3389/feart.2024.1363571</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Earth Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Vegetation extraction in riparian zones based on UAV visible light images and marked watershed algorithm</article-title>
<alt-title alt-title-type="left-running-head">Ma et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/feart.2024.1363571">10.3389/feart.2024.1363571</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Yuanjie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>Xu</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Yaping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2359421/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Information Science and Technology</institution>, <institution>Yunnan Normal University</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Faculty of Geography</institution>, <institution>Yunnan Normal University</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/776242/overview">Jie Cheng</ext-link>, Beijing Normal University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2553674/overview">Quan Zhang</ext-link>, Northwest University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2636537/overview">Xiangchen Meng</ext-link>, Qufu Normal University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Xu Chen, <email>chenxu@ynnu.edu.cn</email>; Yaping Zhang, <email>zhangyp@ynnu.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1363571</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Ma, Chen and Zhang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Ma, Chen and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The riparian zone is an area where land and water are intertwined, and vegetation is rich and complexly distributed. The zone can be directly involved in ecological regulation. In order to protect the ecological environment of the riparian zone, it is necessary to monitor the distribution of vegetation. However, there are many disturbing factors in extracting riparian vegetation, the most serious of which are water bodies with similar colours to the vegetation. To overcome the influence of water bodies on vegetation extraction from UAV imagery of riparian areas, this paper proposes a novel approach that combines the marked watershed algorithm with vegetation index recognition. First, the image is pre-segmented using edge detection, and the output is further refined with the marked watershed algorithm. Background areas are classified as potential regions for vegetation distribution. Subsequently, the final vegetation distribution is extracted from these potential vegetation areas using the vegetation index. The segmentation threshold for the vegetation index is automatically determined using the OTSU algorithm. The experimental results indicate that our method, when applied to UAV aerial imagery of the riparian zone, achieves an overall accuracy of over 94%, a user accuracy of over 97%, and a producer accuracy of over 93%.</p>
</abstract>
<kwd-group>
<kwd>UAV</kwd>
<kwd>riparian zone</kwd>
<kwd>marked watershed algorithm</kwd>
<kwd>vegetation index</kwd>
<kwd>vegetation extraction</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Geoinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Riparian zones are three-dimensional transition zones where terrestrial and aquatic ecosystems directly interact and are among the most biodiverse and productive ecosystems on Earth (<xref ref-type="bibr" rid="B19">Rusn&#xe1;k et al., 2022</xref>). In addition, the riparian zone is referred to as the &#x201c;critical transition zone,&#x201d; which is a conduit for large flows of materials and energy between terrestrial and aquatic ecosystems (<xref ref-type="bibr" rid="B25">Zhang et al., 2019</xref>) and plays an important role in water quality regulation, soil conservation, habitat protection, biodiversity maintenance, climate regulation, riparian landscape, and aquatic ecosystem function. Therefore, riparian zones are key ecosystems within river basins (<xref ref-type="bibr" rid="B25">Zhang et al., 2019</xref>). However, as open freshwater ecosystems at the interface of terrestrial and aquatic realms, riparian zones are less resilient to negative impacts caused by climate change, direct human activities, or artificial demands on water resources (<xref ref-type="bibr" rid="B19">Rusn&#xe1;k et al., 2022</xref>). Moreover, with the increase in urbanization and industrial activities, riparian zones have been severely damaged. According to relevant studies, in Europe, the area of pristine wetlands has been reduced by 80% (<xref ref-type="bibr" rid="B20">Verhoeven, 2014</xref>). In addition, in the Xilin River Basin in Inner Mongolia, due to poor management of reservoirs, the amount of water discharged from the reservoirs is unable to meet the ecological water demand of the downstream riparian zone. In hydrologically wet, average and dry years, only 36%, 19%, and 15% of the ecological water demand can be met, which will directly lead to a reduction in the area of vegetation and a consequent decline in the ecological role of the riparian zone (<xref ref-type="bibr" rid="B6">Duo and Yu, 2020</xref>). Therefore, protecting and restoring riparian zone vegetation is critical to maintaining ecosystem health and promoting sustainable development, and there is a growing need worldwide to protect or restore the ecological health and function of rivers and associated wetlands.</p>
<p>The first task in protecting and restoring riparian zone vegetation is how to extract its vegetation cover area. Remote sensing technology provides a continuous data set from a satellite perspective, which helps to determine the spatial coverage and structural complexity of vegetation and its functioning (<xref ref-type="bibr" rid="B19">Rusn&#xe1;k et al., 2022</xref>). However, most current studies of satellite remote sensing systems are coarse in spatial resolution (e.g., Sentinel 2A (10 m); Landsat TM (30 m); SPOT5 HRV multispectral (10 m)). In their natural state, riparian ecosystems are characterized by a high degree of spatial and temporal heterogeneity (<xref ref-type="bibr" rid="B19">Rusn&#xe1;k et al., 2022</xref>). Therefore, sensors with moderate spatial resolution (&#x3e;4 m &#xd7; 4 m) may not be sufficient to detect and analyze riparian areas because their pixel size often exceeds the physical size of vegetation cover changes in these areas, and thus satellite remote sensing images are insufficient to obtain reliable vegetation measurements (<xref ref-type="bibr" rid="B6">Duo and Yu, 2020</xref>). Unlike satellite remote sensing, unmanned aerial vehicle systems (UAVs) are well suited for riparian zones and riverine ecosystems because of their unprecedented fine scale (<xref ref-type="bibr" rid="B16">M&#xfc;llerov&#xe1; et al., 2021</xref>). UAVs can now be equipped with a variety of sensors, such as visible bands, multispectral, hyperspectral sensors, and Lidar. Among them, RGB images (visible band) are especially widely used due to their convenience, speed, and low price. It has been demonstrated that UAVs deployed with RGB cameras are sufficient to map vegetation cover dynamics (<xref ref-type="bibr" rid="B12">Laslier et al., 2019</xref>). Even more, it is possible to achieve high-accuracy classification of riparian zone vegetation using RGB images instead of multispectral and hyperspectral sensors (<xref ref-type="bibr" rid="B19">Rusn&#xe1;k et al., 2022</xref>). Therefore, the use of aerial drone images for vegetation distribution studies has become a very promising research direction.</p>
<p>Currently, with the development of research, some methods for vegetation cover extraction based on UAV RGB images have emerged, such as extraction methods based on spectral indices and texture information (<xref ref-type="bibr" rid="B12">Laslier et al., 2019</xref>; <xref ref-type="bibr" rid="B8">Gao, et al., 2020</xref>; <xref ref-type="bibr" rid="B11">Kutz et al., 2022</xref>; <xref ref-type="bibr" rid="B22">Xu et al., 2023</xref>), and extraction methods based on machine learning and deep learning (<xref ref-type="bibr" rid="B3">Bhatnagar et al., 2020</xref>; <xref ref-type="bibr" rid="B9">Hamylton et al., 2020</xref>; <xref ref-type="bibr" rid="B17">Onishi and Lse, 2021</xref>; <xref ref-type="bibr" rid="B1">Behera et al., 2022</xref>). Compared to machine learning and deep learning vegetation extraction methods, spectral and texture-based vegetation extraction methods not only do not require training datasets, but are also more computationally efficient and easier to interpret. As a result, they have been widely used in riparian zone processing. For example, <xref ref-type="bibr" rid="B25">Zhang et al. (2019)</xref> proposed a new green-red vegetation index (NGRVI) according to the construction principle of green-red vegetation index (GRVI) and modified green-red vegetation index (MGRVI). The results show that the NGRVI based on UAV visible light images can accurately extract the vegetation information in arid and semi-arid areas, and the extraction accuracy can reach more than 90%. In conclusion, NGRVI can accurately and effectively reflect the vegetation information in arid and semi-arid areas, and become an important technical means for retrieving biological and physical parameters using visible light images. <xref ref-type="bibr" rid="B12">Laslier et al. (2019)</xref> pre-processed UAV images to obtain orthomosaics and calculated vegetation indices, from which texture variables were extracted. Their findings determined that a traditional RGB camera mounted on a UAV was adequate for mapping vegetation cover in the study area. This particular technology demonstrated the feasibility of capturing images and generating information about vegetation cover dynamics at a low cost, given the affordability of RGB cameras. Due to the lack of near-infrared in the visible band, many researchers have resorted to utilizing the spectral reflectance characteristics in the visible band to construct various visible vegetation indices (<xref ref-type="bibr" rid="B25">Zhang et al., 2019</xref>), such as Visible-band Difference Vegetation Index (VDVI) (<xref ref-type="bibr" rid="B21">Wang et al., 2015</xref>), Difference Enhanced Vegetation Index (DEVI) (<xref ref-type="bibr" rid="B26">Zhou et al., 2021</xref>), Normalised Green-Red Difference Index (NGRDI) (<xref ref-type="bibr" rid="B15">Meyer and Neto, 2008</xref>), Modified Green-Red Vegetation Index (MGRVI) (<xref ref-type="bibr" rid="B2">Bendig et al., 2015</xref>), Red-Green-Blue Vegetation Index (RGBVI) and Normalised Green-Blue Difference Index (NGBDI) (<xref ref-type="bibr" rid="B23">Xu et al., 2017</xref>). However, because visible vegetation indices can only determine the type of ground cover by its colour, it is difficult to distinguish similarly coloured features using only simple indices. For example, if a clear body of water is more than 2 m deep and is surrounded by abundant vegetation, the reflection and refraction of light can cause the water&#x2019;s colour to change to a vegetative green. In addition, the presence of algae and plankton in a body of water can cause aerial images taken by a drone to appear darker green, especially in urban areas. The high-resolution imagery captured by UAVs enables us to acquire detailed information on vegetation distribution within the riparian zone, facilitating real-time monitoring and assessment of its ecological health status. However, given that the riparian zone is a complex ecosystem, the vegetation distribution is intricately linked to the diverse environments of rivers and lakes. Moreover, the close color proximity of water bodies and vegetation poses a significant challenge in accurately identifying vegetation from UAV aerial imagery. Water bodies are often misidentified as vegetation when detected using the Visible Band Vegetation Index (VBVI). Meanwhile, due to the complexity of the riparian environment, high-resolution images acquired by UAVs often have various disturbances, such as waves, shadows, sunlight reflections, etc., in the water area, which further increases the difficulty of vegetation extraction. Therefore, it is necessary to incorporate other information (e.g., texture information) to jointly delineate and extract the vegetation cover area. Unfortunately, there is still a lack of research on efficiently extracting vegetation in complex riparian zones from high-resolution RGB images from UAVs.</p>
<p>Based on observing and analyzing the UAV RGB images of the riparian zone, this study proposes a novel method for vegetation extraction in the riparian zone by combining the marked watershed algorithm and vegetation index recognition. The basic ideas of this method are to 1) distinguish water bodies and non-water bodies by texture. 2) Discriminate between vegetation and non-vegetation by vegetation index after excluding water bodies. Unlike the traditional marked watershed algorithm, this method is characterized by obtaining the potential vegetation distribution area through the background marker of the marked watershed algorithm, which avoids the problem of over-segmentation. Specifically, the method extracts the texture information in RGB images by the Canny operator and then marks the potential complex riparian vegetation cover area by a marked watershed algorithm. Finally, the vegetation cover area is extracted by setting a threshold through the visible light vegetation index. The experiments prove that the method can efficiently and accurately extract the vegetation in complex riparian zones, and effectively solve the problem that it is difficult to distinguish between green water bodies and vegetation by visible light vegetation index. This study will provide practical technical support for sustainable ecological restoration and management of riparian zones. In order to show the effectiveness of the proposed method, six UAV aerial images of the riparian zone with different regions were selected for experimental demonstration.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Materials</title>
<p>In this study, a DJI Phantom 4 RTK drone was used for data acquisition and captured UAV remote sensing images in R, G, and B bands with a size of 5472&#x2a;3648 pixels. The drone images were captured at a relative flight height of 120 m, with a Ground Sample Distance (GSD) of 3.5 cm/px. The images were taken on 21 July 2023, between 10 a.m. and 4 p.m. In order to illustrate the validity and generalizability of the experimental results, we selected six strongly representative UAV aerial images of riparian zones with different surface features and water conditions for the experimental study, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. For example, the water surface is calm, the water body is uniform in color, and the riverbank is almost completely covered by vegetation (<xref ref-type="fig" rid="F1">Figure 1A</xref>). The water body has different depths, resulting in the uneven color of the water body and sunlight reflection, and there are some floating objects on the water surface and some bare soil and roads in the vegetation-covered area on the river bank (<xref ref-type="fig" rid="F1">Figure 1B</xref>). There is visible floating debris and trash on the water, shadows obscure part of the water surface, and power lines are passing over the water surface, causing obscuration (<xref ref-type="fig" rid="F1">Figure 1C</xref>). Vegetation is abundant, and sunlight reflects off the water surface, creating pronounced ripples (<xref ref-type="fig" rid="F1">Figure 1D</xref>). The water surface exhibites ripple distribution and the water body&#x2019;s depth varies (<xref ref-type="fig" rid="F1">Figure 1E</xref>). The presence of algae causes uneven green coloration on the water surface (<xref ref-type="fig" rid="F1">Figure 1F</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Aerial drone images of the complex riparian zone: <bold>(A)</bold> Region I, <bold>(B)</bold> Region II, <bold>(C)</bold> Region III, <bold>(D)</bold> Region IV, <bold>(E)</bold> Region V, <bold>(F)</bold> Region VI.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Methods</title>
<sec id="s2-2-1">
<title>2.2.1 Algorithm implementation process and flowchart</title>
<p>The implementation process of the vegetation extraction method proposed in this paper consists of seven steps, as depicted in <xref ref-type="fig" rid="F2">Figure 2</xref>. The pseudo-code of the vegetation extraction process is presented in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>, and the algorithm is implemented using Python and OpenCV.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Implementation process of the vegetation extraction method.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g002.tif"/>
</fig>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Vegetation extraction.<list list-type="simple">
<list-item>
<p>
<bold>Input</bold>: UAV RGB image</p>
</list-item>
<list-item>
<p>
<bold>Output</bold>: Vegetation coverage area in the image</p>
</list-item>
<list-item>
<p>1&#x2003;Perform edge detection and vegetation index calculation for the input image to obtain images <italic>I</italic>
<sub>1</sub> and <italic>I</italic>
<sub>v</sub>, respectively;</p>
</list-item>
<list-item>
<p>2&#x2003;The image <italic>I</italic>
<sub>1</sub> was binarised, and Its areas of high-frequency texture (potential vegetation) were set as the background, followed by a morphological closing operation to eliminate holes and then an opening operation to reduce noise to obtain the image <italic>I</italic>
<sub>2</sub>;</p>
</list-item>
<list-item>
<p>3&#x2003;The image <italic>I</italic>
<sub>2</sub> was subjected to distance transformation to obtain the image <italic>I</italic>
<sub>3</sub>;</p>
</list-item>
<list-item>
<p>4&#x2003;Perform threshold segmentation and binarization for the distance transformation image <italic>I</italic>
<sub>3</sub> to obtain the watershed seed area image <italic>I</italic>
<sub>4</sub>;</p>
</list-item>
<list-item>
<p>5&#x2003;The image <italic>I</italic>
<sub>2</sub> subtracts the image <italic>I</italic>
<sub>4</sub> to obtain uncertainty areas;</p>
</list-item>
<list-item>
<p>6&#x2003;Perform marked watershed algorithm with seed areas, uncertainty areas, and UAV RGB image as inputs to extract the potential vegetation areas;</p>
</list-item>
<list-item>
<p>7&#x2003;Perform threshold segmentation for potential vegetation areas with vegetation indices to obtain final vegetation cover areas.</p>
</list-item>
</list>
</p>
</statement>
</p>
<p>By performing &#x201c;Distance Transformation,&#x201d; the distance from foreground pixels to the nearest background pixels can be calculated, resulting in a grayscale image where higher grayscale levels indicate greater distances from the background. The &#x201c;seed area&#x201d; is generated through threshold segmentation and binarization of the &#x201c;distance transformation&#x201d; image. By subtracting the &#x201c;seed area&#x201d; from image <italic>I</italic>
<sub>2</sub> described in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>, we obtain the &#x201c;uncertainty area&#x201d; image. In the marking watershed algorithm, diffusing the &#x201c;seed area&#x201d; towards the &#x201c;uncertainty area&#x201d; results in the segmentation outcome. Since all vegetation in these results is located within the background region, this area is identified as the &#x201c;Potential Vegetation area&#x201d;.</p>
<p>It is important to note that, in order to present the processing results more clearly, we have utilized the appropriate color mode for display. In image <italic>I</italic>
<sub>3</sub>, the color change from red to blue represents the distance between the foreground pixels and the background pixels, and the red color indicates the distance is further away; in image <italic>I</italic>
<sub>4</sub>, the light blue on the top and the yellow and red on the left local zoom represent different seed areas, and the dark blue in the middle represents the background. In image <italic>I</italic>
<sub>5</sub>, the red color represents the uncertain areas; in image <italic>I</italic>
<sub>6</sub>, the dark blue area labeled &#x201c;PVA&#x201d; is the potential vegetation area.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Vegetation index</title>
<p>Vegetation index is a technique developed from remote sensing. By analyzing the differences in the different spectral curves presented by different objects in images taken by multi/hyperspectral satellites, data in specific bands can be combined to highlight the representation of specific object characteristics. This technique can distinguish between object types or related object characteristics, such as plant or vegetation abundance, in remotely sensed images. As there are more types of vegetation indices, they should be selected according to the actual situation, such as remote sensing image type, band composition, land cover type, vegetation type, etc., to achieve more accurate extraction results. For general RGB images, vegetation indices containing only visible bands should be selected. Common visible band vegetation indices are shown in Eqs <xref ref-type="disp-formula" rid="e1">1</xref>&#x2013;<xref ref-type="disp-formula" rid="e6">6</xref>:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn mathvariant="bold">2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn mathvariant="bold">2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">E</mml:mi>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mn mathvariant="bold">3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mn mathvariant="bold">3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mn mathvariant="bold">3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="bold-italic">N</mml:mi>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi mathvariant="bold-italic">R</mml:mi>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="bold-italic">M</mml:mi>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi mathvariant="bold-italic">R</mml:mi>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:mi mathvariant="bold-italic">R</mml:mi>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi mathvariant="bold-italic">B</mml:mi>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="bold-italic">N</mml:mi>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi mathvariant="bold-italic">B</mml:mi>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">g</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3c1;</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>In the equations: <italic>&#x3c1;</italic>
<sub>
<italic>r</italic>
</sub>, <italic>&#x3c1;</italic>
<sub>
<italic>g</italic>
</sub>, <italic>&#x3c1;</italic>
<sub>
<italic>b</italic>
</sub> are the image pixel values of the red, green and blue 3 bands respectively.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 OTSU image segmentation</title>
<p>The OTSU method (<xref ref-type="bibr" rid="B18">Otsu, 1979</xref>) is an algorithm used to determine optimal segmentation thresholds for image binarization. This algorithm automatically calculates the thresholds based on the characteristics of the image (<xref ref-type="bibr" rid="B24">Yang et al., 2014</xref>). The specific calculation process is shown in Eq. <xref ref-type="disp-formula" rid="e7">7</xref>:<disp-formula id="e7">
<mml:math id="m7">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m8">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the inter-class variance, <inline-formula id="inf2">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf3">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the probability and mean of class <italic>i</italic>, respectively.</p>
<p>The OTSU method, also known as the maximum inter-class variance method, offers several advantages: it involves simple calculations, is convenient to use, and is unaffected by the image&#x2019;s brightness. Additionally, it enables fast segmentation in a simple bimodal scene. However, there are also some disadvantages to consider. The method is susceptible to noise interference, and when the image lacks distinct bimodal peaks, accurate segmentation cannot be achieved. Furthermore, its effectiveness diminishes when applied to images with multiple peaks.</p>
</sec>
<sec id="s2-2-4">
<title>2.2.4 Canny operator</title>
<p>The Canny operator is an exact edge detection algorithm known for its robust resistance to interference. It incorporates dual thresholds and multi-level characteristics, which make it adaptable to complex images. The fundamental principle of Canny edge detection involves converting an image to grayscale and identifying edges by detecting significant variations in grey values (<xref ref-type="bibr" rid="B5">Ding and Goshtasby, 2001</xref>; <xref ref-type="bibr" rid="B14">McIlhagga, 2011</xref>). This is based on the observation that changes in luminance typically occur at the edges of the image. Mathematically, this can be achieved by calculating the first-order partial derivatives, where points with extremely large partial derivatives represent the edges of the image (<xref ref-type="bibr" rid="B13">Liu and Jezek, 2004</xref>), the specific calculation process is shown in Eq. <xref ref-type="disp-formula" rid="e8">8</xref>.<disp-formula id="e8">
<mml:math id="m11">
<mml:mrow>
<mml:mo>&#x2207;</mml:mo>
<mml:mi mathvariant="bold-italic">f</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:mi mathvariant="bold-italic">f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:mi mathvariant="bold-italic">f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:mi mathvariant="bold-italic">y</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi mathvariant="bold-italic">y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>Where <italic>G</italic>
<sub>
<italic>x</italic>
</sub>, <italic>G</italic>
<sub>
<italic>y</italic>
</sub> denote the first-order partial derivatives at the point (<italic>x,y</italic>) in the image <italic>f</italic>, respectively.</p>
</sec>
<sec id="s2-2-5">
<title>2.2.5 Marked watershed algorithm</title>
<p>The Watershed algorithm belongs to the category of image segmentation algorithms. Its principle involves mapping the same gray levels in an image to contour lines in geography, creating a topographic surface defined by the gray values of the image. Basins are formed in areas with extremely low gray values (<xref ref-type="bibr" rid="B10">Kornilov &#x26; Safonov, 2018</xref>). If we imagine water flooding the surface from the lowest point, dams are built to prevent adjacent basins from merging. These dams, known as watershed lines, serve as the segmentation lines for the image. The watershed algorithm has several advantages, including its simplicity, intuitiveness, and potential for parallelization. However, it has a significant drawback, which is over-segmentation caused by numerous local minima in the image. To mitigate the severe over-segmentation issue, the marked watershed algorithm has been proposed.</p>
<p>In the marked watershed algorithm, the foreground and the background are automatically determined by integrating techniques like edge detection, binarization, and morphological operations. Then, a distance transformation is conducted to generate the watershed seed areas. However, since vegetation texture tends to be more intricate, while water bodies exhibit relatively uniform and cohesive textures, we designate the water bodies as the foreground to obtain smoother segmentation boundaries in vegetation extraction.</p>
</sec>
<sec id="s2-2-6">
<title>2.2.6 Kappa coefficient</title>
<p>Classification accuracy is also calculated using the Kappa coefficient (<xref ref-type="bibr" rid="B7">Foody, 2002</xref>), the specific calculation process is shown in Eq. <xref ref-type="disp-formula" rid="e10">9</xref>.<disp-formula id="e10">
<mml:math id="m12">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mi mathvariant="bold-italic">p</mml:mi>
<mml:mi mathvariant="bold-italic">p</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">P</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">P</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">P</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>o</italic>
</sub> represents the agreement between predicted results and actual results, which can be calculated using a confusion matrix. <italic>P</italic>
<sub>
<italic>e</italic>
</sub> refers to the agreement between predicted results and actual results when the predictions are made randomly. As the Kappa coefficient approaches 1, it signifies an increased degree of concordance in classification. When approaching 0, it suggests that classification performance is equivalent to random predictions. However, once it approaches &#x2212;1, it indicates reduced consistency in classification compared to random predictions.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Extracting vegetation cover using only vegetation indices</title>
<p>In <xref ref-type="fig" rid="F1">Figure 1</xref>, the green vegetation information of each region appears to be highly similar. To select an optimal threshold for vegetation recognition, we plotted the histogram corresponding to the VDVI and identified the lowest value of 0.11 between the double peaks. This threshold has proven to be more accurate in extracting vegetation information from our images. Consequently, we have chosen to use this threshold consistently throughout our subsequent VDVI-based vegetation extraction processes.</p>
<p>The results of threshold segmentation after calculating the VDVI vegetation index for each subplot in <xref ref-type="fig" rid="F1">Figure 1</xref> are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. As can be seen from the figure, all subplots have water bodies that are difficult to completely separate from the vegetation. Because the color of the water body is similar to the vegetation, setting a larger segmentation threshold can further separate the water body. However, it will also lead to a large amount of vegetation loss.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The results of extracting vegetation cover using only vegetation indices (green indicates areas identified as vegetation, while black indicates areas identified as non-vegetation): <bold>(A)</bold> Region I, <bold>(B)</bold> Region II, <bold>(C)</bold> Region III, <bold>(D)</bold> Region IV, <bold>(E)</bold> Region V, <bold>(F)</bold> Region VI.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g003.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Extracting vegetation cover using only texture information</title>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> shows the threshold segmentation results of the Canny edge detection for each subplot shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. It can be seen that all subplots can effectively separate water bodies, but they cannot completely exclude the non-vegetated surface cover. Additionally, the most serious issue is that a large number of voids appear within the vegetation cover. The Otsu algorithm is utilized to automatically determine the threshold value in this method. While setting a larger segmentation threshold can help in further separating the non-vegetated surface cover, it comes at the cost of potentially losing a significant amount of vegetation and creating larger voids within the vegetation cover.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The results of extracting vegetation cover using canny edge detection only: <bold>(A)</bold> Region I, <bold>(B)</bold> Region II, <bold>(C)</bold> Region III, <bold>(D)</bold> Region IV, <bold>(E)</bold> Region V, <bold>(F)</bold> Region VI.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g004.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 Extraction of vegetation cover using our method</title>
<p>The results of extracting the vegetation cover of each subplot in <xref ref-type="fig" rid="F1">Figure 1</xref> using the method proposed in this paper are shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. It can be seen from the figure that all the water bodies, which can easily be confused with vegetation, have been excluded completely in all the subplots, while the vegetation has been retained to the maximum extent.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The results of extracting vegetation cover using our method: <bold>(A)</bold> Region I, <bold>(B)</bold> Region II, <bold>(C)</bold> Region III, <bold>(D)</bold> Region IV, <bold>(E)</bold> Region V, <bold>(F)</bold> Region VI.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g005.tif"/>
</fig>
<p>To quantitatively assess the accuracy of vegetation extraction, water bodies were manually labeled and removed from each subplot in <xref ref-type="fig" rid="F1">Figure 1</xref>. The VDVI vegetation index was then calculated, and the vegetation was extracted. The extraction results were compared with those obtained by the method in this paper, and the confusion matrix was calculated to obtain the overall accuracy, producer accuracy, and user accuracy, as shown in <xref ref-type="table" rid="T1">Table 1</xref>. The table shows that the overall accuracy of all regions is above 95% except for region (c), where the overall accuracy is slightly below 95%. The user accuracy is very high in all regions, while the producer accuracy is slightly lower. Among them, the producer accuracy of region (c) is slightly lower than 90% because there is relatively less vegetation cover in the region (c), and more dark shadows cover the vegetation. This causes the edge detection algorithm to have lower values in these areas, which are swamped and marked as the foreground in the watershed algorithm, resulting in non-vegetated areas.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Accuracy after interference removal using our method.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Regions</th>
<th align="center">Overall accuracy/%</th>
<th align="center">User&#x2019;s accuracy/%</th>
<th align="center">Producer&#x2019;s accuracy/%</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">(a)</td>
<td align="center">97.25</td>
<td align="center">99.93</td>
<td align="center">96.02</td>
</tr>
<tr>
<td align="center">(b)</td>
<td align="center">99.27</td>
<td align="center">99.24</td>
<td align="center">99.26</td>
</tr>
<tr>
<td align="center">(c)</td>
<td align="center">94.95</td>
<td align="center">99.20</td>
<td align="center">88.66</td>
</tr>
<tr>
<td align="center">(d)</td>
<td align="center">95.49</td>
<td align="center">99.85</td>
<td align="center">94.04</td>
</tr>
<tr>
<td align="center">(e)</td>
<td align="center">98.89</td>
<td align="center">99.92</td>
<td align="center">98.41</td>
</tr>
<tr>
<td align="center">(f)</td>
<td align="center">99.68</td>
<td align="center">99.78</td>
<td align="center">99.72</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4">
<title>3.4 Vegetation extraction for public datasets with ground truth using our method</title>
<p>The Urban Drone Dataset (UDD) (<xref ref-type="bibr" rid="B4">Chen et al., 2018</xref>) was collected using the DJI Phantom 4 at heights ranging from 60 to 100 m. The image size is 3000 &#xd7; 4000 or 4096 &#xd7; 2160. This dataset is divided into three types: UDD3, UDD5, and UDD6, each representing different scenarios. To validate the accuracy of our method, we selected three images (<xref ref-type="fig" rid="F6">Figures 6A&#x2013;C</xref>) from UDD5 that resemble the riparian zone environment, and their corresponding ground truth labels were provided (<xref ref-type="fig" rid="F6">Figures 6D&#x2013;F</xref>). The results of extracting the vegetation cover of each subplot in <xref ref-type="fig" rid="F6">Figures 6A&#x2013;C</xref> using the method proposed in this paper are shown in <xref ref-type="fig" rid="F6">Figures 6G&#x2013;I</xref>. The extraction results are compared with the true values of the images, and the extraction accuracy is calculated, as shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The Urban Drone Dataset image and corresponding vegetation extraction results, <bold>(A&#x2013;C)</bold>: original images of three different regions, <bold>(D&#x2013;F)</bold>: labeled image corresponding to the original image, where green indicates vegetation, and other colors represent non-vegetation, <bold>(G&#x2013;I)</bold>: The results of extracting vegetation cover using our method.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g006.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>VDVI extraction accuracy after eliminating interference.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Regions</th>
<th align="center">Overall accuracy/%</th>
<th align="center">User&#x2019;s accuracy/%</th>
<th align="center">Producer&#x2019;s accuracy/%</th>
<th align="center">Kappa</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">
<xref ref-type="fig" rid="F6">Figure 6A</xref>
</td>
<td align="center">90.17</td>
<td align="center">98.38</td>
<td align="center">70.36</td>
<td align="center">0.76</td>
</tr>
<tr>
<td align="center">
<xref ref-type="fig" rid="F6">Figure 6B</xref>
</td>
<td align="center">92.31</td>
<td align="center">97.85</td>
<td align="center">57.83</td>
<td align="center">0.69</td>
</tr>
<tr>
<td align="center">
<xref ref-type="fig" rid="F6">Figure 6C</xref>
</td>
<td align="center">81.65</td>
<td align="center">97.85</td>
<td align="center">59.70</td>
<td align="center">0.61</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From <xref ref-type="fig" rid="F6">Figure 6</xref>, it can be seen that the original image in UDD5 is darker and has more shadows and reflections. On the other hand, from <xref ref-type="table" rid="T2">Table 2</xref>, the overall accuracy and kappa coefficient of classification of this paper&#x2019;s method for <xref ref-type="fig" rid="F6">Figures 6A&#x2013;C</xref> are 90.17%, 92.31%, 81.65%, and 0.76, 0.69, 0.61, respectively. However, the producer accuracy is lower, 70.36%, 57.83%, and 59.7% in <xref ref-type="fig" rid="F6">Figures 6A&#x2013;C</xref>, respectively. <xref ref-type="fig" rid="F6">Figure 6B</xref> has the highest Overall Accuracy, but it has a lower Kappa than <xref ref-type="fig" rid="F6">Figure 6A</xref> due to its uneven distribution of error pixels. Specifically, compared to <xref ref-type="fig" rid="F6">Figures 6A,B</xref> has fewer errors in recognizing water body pixels as vegetation pixels, leading to a slightly higher Overall Accuracy. The uniformity of the error pixel distribution directly influences the Kappa coefficient. <xref ref-type="fig" rid="F6">Figure 6B</xref> has fewer error pixels, and the distribution location of these pixels is not uniform enough, resulting in a lower Kappa coefficient for <xref ref-type="fig" rid="F6">Figure 6B</xref> than for <xref ref-type="fig" rid="F6">Figure 6A</xref>.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<sec id="s4-1">
<title>4.1 Choice of edge detection operator</title>
<p>The Canny and Sobel operators are widely used algorithms for image edge detection. The Canny operator determines image edges by considering the direction and magnitude of the gradient, while the Sobel operator calculates the gradient using pixel value differences (simple subtraction of two pixels). As a result, the Canny operator can preserve edge continuity more effectively during edge detection, resulting in more detailed and clearer detected edges, especially in scenes that require high-precision edge detection, such as image recognition. On the other hand, the Sobel operator may introduce significant errors when processing right-angled edges. However, it is faster than the Canny operator, making it more suitable for real-time applications requiring higher speed, such as video surveillance.</p>
<p>In order to find edge detection operators suitable for the characteristics of riparian zone images, in this study, we considered the Canny operator and Sobel operator, which are outstanding in edge recognition accuracy and noise point sensitivity. We converted <xref ref-type="fig" rid="F1">Figures 1A&#x2013;C</xref> from RGB images to grayscale images and then utilized Canny and Sobel edge detection operators and marked the areas in the images where segmenting the boundaries along the water body was challenging using red boxes, as illustrated in <xref ref-type="fig" rid="F7">Figure 7</xref>. These challenging regions exhibit characteristics such as shallow water bodies near the shore, with a color that appears closer to the shore, and contain various undesirable elements, including scum, garbage, and power lines. All these interferences directly affect the edge detection effect of the two operators. However, it is evident that Canny detection retains more edge information than Sobel detection and provides a more refined detection. Therefore, we chose the Canny operator as the pre-extraction segmentation algorithm before the detection of the watershed algorithm.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The results using different edge detection operators: <bold>(A)</bold>, <bold>(B)</bold>, and <bold>(C)</bold> correspond to the detection results using the Canny operator for areas depicted in <xref ref-type="fig" rid="F1">Figures 1A&#x2013;C</xref>, respectively; <bold>(D)</bold>, <bold>(E)</bold>, and <bold>(F)</bold> correspond to the detection results using the Sobel operator for areas shown in <xref ref-type="fig" rid="F1">Figures 1A&#x2013;C</xref>, respectively.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g007.tif"/>
</fig>
<p>The watershed algorithm was used to segment the results of Canny and Sobel edge detection for <xref ref-type="fig" rid="F1">Figures 1A&#x2013;C</xref>, as shown in <xref ref-type="fig" rid="F8">Figure 8</xref>. It can be seen from the images that the segmentation results based on the Canny operator are more accurate. As indicated by the yellow boxes in the images, we can observe that for <xref ref-type="fig" rid="F1">Figure 1B</xref>, the segmentation result based on the Canny operator successfully separated non-water areas locally (<xref ref-type="fig" rid="F8">Figure 8B</xref>), while the result from the Sobel operator did not (<xref ref-type="fig" rid="F8">Figure 8E</xref>). Similarly, for <xref ref-type="fig" rid="F1">Figure 1C</xref>, the segmentation result based on the Canny operator segmented water areas locally (<xref ref-type="fig" rid="F8">Figure 8C</xref>), while the result from the Sobel operator did not (<xref ref-type="fig" rid="F8">Figure 8F</xref>). Comparing <xref ref-type="fig" rid="F1">Figure 1</xref> and <xref ref-type="fig" rid="F8">Figure 8</xref>, we can see that although <xref ref-type="fig" rid="F8">Figure 8</xref> shows some oversegmentation, water bodies and potential vegetation were separated into different colour areas. Therefore, after extracting the potential vegetation areas, vegetation extraction can be achieved using vegetation indices.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>The results of watershed segmentation based on different edge detection operators, <bold>(A&#x2013;C)</bold>: based on Canny edge detection, <bold>(D&#x2013;F)</bold>: based on Sobel edge detection.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g008.tif"/>
</fig>
</sec>
<sec id="s4-2">
<title>4.2 Choice of visible light vegetation indices</title>
<p>Several visible light vegetation indices are currently proposed, depending on the purpose of the application. Each of these indices has advantages and disadvantages in distinguishing land cover types. In order to select the most appropriate index, six visible light indices were selected for testing in this paper. First, a UDD5 scene image (<xref ref-type="fig" rid="F6">Figure 6A</xref>) was selected. After removing the water body interference using the method in this paper, the potential vegetation areas were extracted by vegetation segmentation using VDVI, DEVI, RGBVI, MGRVI, NGRDI, and NGBDI, respectively. Following the vegetation extraction approach used in <xref ref-type="fig" rid="F6">Figure 6A</xref>, we also processed <xref ref-type="fig" rid="F6">Figure6B,C</xref>. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, for different vegetation indices, we calculated the average of each evaluation metric across the three images.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Mean extraction accuracy of different vegetation indices.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Vegetation index</th>
<th align="center">Overall accuracy/%</th>
<th align="center">User&#x2019;s accuracy/%</th>
<th align="center">Producer&#x2019;s accuracy/%</th>
<th align="center">Kappa/%</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">VDVI</td>
<td align="center">86.75</td>
<td align="center">98.16</td>
<td align="center">57.94</td>
<td align="center">64.23</td>
</tr>
<tr>
<td align="center">DEVI</td>
<td align="center">82.94</td>
<td align="center">99.48</td>
<td align="center">44.95</td>
<td align="center">52.48</td>
</tr>
<tr>
<td align="center">RGBVI</td>
<td align="center">85.99</td>
<td align="center">98.69</td>
<td align="center">55.50</td>
<td align="center">62.13</td>
</tr>
<tr>
<td align="center">MGRVI</td>
<td align="center">84.49</td>
<td align="center">82.50</td>
<td align="center">63.78</td>
<td align="center">60.64</td>
</tr>
<tr>
<td align="center">NGRDI</td>
<td align="center">84.56</td>
<td align="center">80.78</td>
<td align="center">68.00</td>
<td align="center">61.77</td>
</tr>
<tr>
<td align="center">NGBDI</td>
<td align="center">85.49</td>
<td align="center">98.18</td>
<td align="center">55.18</td>
<td align="center">61.62</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As can be seen from <xref ref-type="table" rid="T3">Table 3</xref>, VDVI has the highest accuracy, with the overall accuracy and kappa coefficient reaching 86.75% and 0.6423, respectively. Secondly, for the sub-map with relatively complex land cover types (<xref ref-type="fig" rid="F1">Figure 1C</xref>), <xref ref-type="fig" rid="F9">Figure 9</xref> presents the histograms of the six visible vegetation indices of the potentially vegetated areas after removing the water body using this method.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Histograms of different vegetation indices: <bold>(A)</bold> VDVI, <bold>(B)</bold> DEVI, <bold>(C)</bold> NGRDI, <bold>(D)</bold> MGRVI, <bold>(E)</bold> RGBVI, <bold>(F)</bold> NGBDI for <xref ref-type="fig" rid="F1">Figure 1C</xref>).</p>
</caption>
<graphic xlink:href="feart-12-1363571-g009.tif"/>
</fig>
<p>As can be seen in <xref ref-type="fig" rid="F9">Figure 9</xref>, although the histogram distributions of the different vegetation indices varied greatly, they basically maintained a bimodal distribution. This ensures the correctness of the subsequent segmentation of vegetation and non-vegetation using the OTSU method. The optimal segmentation thresholds for subplots from (a) to (f) in <xref ref-type="fig" rid="F9">Figure 9</xref> are 0.11, 0.88, 0.07, 0.13, 0.2, and 0.06, respectively. Among them, the histogram distribution of the VDVI indices satisfies both the undulating and smooth requirements, leading to more accurate segmentation results.</p>
</sec>
<sec id="s4-3">
<title>4.3 Selection of <italic>P</italic> parameter values in marked watershed algorithm</title>
<p>In <xref ref-type="fig" rid="F2">Figure 2</xref>, the maximum value resulting from the Distance Transformation is multiplied by a percentage value (denoted as <italic>P</italic>), and this product is set as a threshold. This threshold is then used for threshold segmentation of the Distance Transformation image <italic>I</italic>
<sub>3</sub> to obtain the Seed Areas image <italic>I</italic>
<sub>4</sub>. By adjusting the value of <italic>P</italic>, we can influence the number of Seed Areas and, consequently, the segmentation results that follow. The marked watershed algorithm utilizes the <italic>P</italic> parameter value to regulate the smoothness of boundaries and the fineness of segmentation results. Higher <italic>p</italic> values produce smoother boundaries and more consistent area segmentation, while lower <italic>p</italic> values generate more detailed but less consistent boundaries. As a key factor in determining the efficacy and outcome of the algorithm, the <italic>P</italic> parameter value significantly impacts the overall performance. For instance, <xref ref-type="fig" rid="F10">Figures 10A&#x2013;C</xref> presents an example where <italic>P</italic> is systematically increased at 0.2 intervals from 0.3 for <xref ref-type="fig" rid="F1">Figure 1B</xref>. The figures clearly illustrate the decrease in the number of seed areas with an increase in <italic>P</italic>. However, too few seed areas may result in certain water bodies being overlooked in the segmentation process, leading to their inclusion in the potential vegetation areas. This, in turn, may cause errors in the final vegetation segmentation, as depicted in the red-boxed portion of <xref ref-type="fig" rid="F10">Figures 10D&#x2013;F</xref>.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>The effects of different <italic>p</italic> values for <xref ref-type="fig" rid="F1">Figure 1B</xref>: Specifically, <bold>(A)</bold> and <bold>(D)</bold> correspond to <italic>P</italic> equaling 0.3, <bold>(B)</bold> and <bold>(E)</bold> represent <italic>P</italic> equaling 0.5, while <bold>(C)</bold> and <bold>(F)</bold> depict <italic>P</italic> equaling 0.7.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g010.tif"/>
</fig>
</sec>
<sec id="s4-4">
<title>4.4 Defects and deficiencies&#x2014;special textures and shadows</title>
<p>The segmentation process, which involves distinguishing water body areas from potential vegetation areas, relies heavily on the disparity between the textures of water bodies and vegetation. Consequently, areas with water bodies exhibiting similar textures to vegetation textures will likely be erroneously classified as potential vegetation areas during the segmentation. Additionally, if such areas also exhibit high values for the VDVI index, it becomes challenging to differentiate them from actual vegetation using the approach outlined in this paper. This challenge is demonstrated by the red and yellow boxed sections in <xref ref-type="fig" rid="F11">Figures 11A, B</xref> and the white boxes in <xref ref-type="fig" rid="F11">Figures 11C, D</xref>. However, if the water body areas do not exhibit high VDVI values, the method outlined in this paper can still successfully remove these areas at the OTSU segmentation stage, as shown in the blue box portion of <xref ref-type="fig" rid="F11">Figures 11E, F</xref>.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Vegetation extraction results for different water colours and ripples, <bold>(A&#x2013;D)</bold>: High-strength water ripple, <bold>(E F)</bold>: Low-strength water ripple.</p>
</caption>
<graphic xlink:href="feart-12-1363571-g011.tif"/>
</fig>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> demonstrates that the method proposed in this paper achieves a high overall accuracy and a high user accuracy, implying that our approach has fewer instances of misclassification. However, the lower producer accuracy suggests that the method occasionally fails to correctly classify certain categories, as illustrated in <xref ref-type="fig" rid="F6">Figure 6</xref>. This discrepancy can be attributed primarily to the time and angle at which the image was captured. As can be seen in <xref ref-type="fig" rid="F6">Figure 6</xref>, the images of the UDD5 dataset were basically taken in the evening with insufficient light, while the sun was tilted at a large angle. The low illumination condition (C) in <xref ref-type="fig" rid="F6">Figure 6</xref> was utilized for vegetation extraction with the method described in this paper, and the results are depicted in <xref ref-type="fig" rid="F12">Figure 12</xref>, this lighting condition results in darker shadows on vegetation (indicated by the red box) and shaded water bodies (indicated by the yellow box). These darker shadows cause a loss of textural information in the areas where the watershed algorithm cannot flood, and as such, they cannot be excluded from the segmentation process (as indicated by the yellow box). Additionally, these darker shadows significantly lower VDVI values, resulting in a greater loss of vegetation categories (as indicated by the red box). However, these problems can be easily overcome by selecting a clear and cloudless midday period for data acquisition, which avoids low light and strong shadows, thus achieving high segmentation accuracy.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>Effect of shadows and reflections on vegetation extraction: <bold>(A)</bold> Original image, <bold>(B)</bold> Vegetation extraction results of the method in this paper..</p>
</caption>
<graphic xlink:href="feart-12-1363571-g012.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this study, we have introduced a straightforward and efficient approach capable of automatically extracting riparian zone vegetation from aerial UAV images containing only visible bands despite the numerous disturbing factors encountered in complex riparian zones. Some of these factors include shallows, green algae, flotsam, sunlight reflections, and tree shadows. Specifically:<list list-type="simple">
<list-item>
<p>1 The method proposed in this paper presents a novel idea to solve the challenge of vegetation extraction in riparian zones and can achieve this goal solely based on the visible band vegetation index. This approach has significant practical applications.</p>
</list-item>
<list-item>
<p>2 Although the method proposed in this paper demonstrates some ability to mitigate factors that interfere with vegetation extraction, further research is required to address other characteristic interferences that are similar to vegetation. This includes optimizing the method for scenarios characterized by high-intensity ripple interference.</p>
</list-item>
<list-item>
<p>3 Regarding the accuracy assessment, the dense vegetation and high image resolution in the study area pose challenges in manually labeling images to obtain ground-truth data. This complexity may introduce limitations to the accuracy of the truth-value images used in our accuracy validation. Consequently, future research should aim to further investigate this issue, with the goal of acquiring more accurate vegetation distribution maps of the riparian zone to serve as reference data.</p>
</list-item>
<list-item>
<p>4 Currently, research on vegetation extraction has expanded to the field of deep learning. However, there is still a lack of deep learning techniques for semantic segmentation of riparian zone vegetation due to the challenges in manually labeling images for training. The proposed method in this article can provide helpful vegetation labels for riparian zone images, addressing the issue of insufficient existing datasets and serving as a reference for future research on deep learning-based vegetation extraction.</p>
</list-item>
</list>
</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>YM: Data curation, Formal Analysis, Software, Validation, Visualization, Writing&#x2013;original draft. XC: Conceptualization, Formal Analysis, Funding acquisition, Investigation, Methodology, Software, Supervision, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. YZ: Investigation, Project administration, Resources, Supervision, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was funded by two sources. The first source of funding was the Yunnan Provincial Agricultural Basic Research Joint Special Project, supported by the Yunnan Provincial Science and Technology Department, with Grant No. 202101BD070001-042. The second source of funding came from the Yunnan Ten-thousand Talents Program, supported by the Yunnan Provincial Department of Human Resources and Social Security.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Behera</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Bakshi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sa</surname>
<given-names>P. K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Vegetation extraction from UAV-based aerial images through deep learning</article-title>. <source>Comput. Electron. Agric.</source> <volume>198</volume>, <fpage>107094</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2022.107094</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bendig</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Aasen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bolten</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bennertz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Broscheit</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Combining UAV-based plant height from crop surface models, visible, and near infrared vegetation indices for biomass monitoring in barley</article-title>. <source>Int. J. Appl. Earth Observation Geoinformation</source> <volume>39</volume>, <fpage>79</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2015.02.012</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhatnagar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gill</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Drone image segmentation using machine and deep learning for mapping raised bog vegetation communities</article-title>. <source>remote Sens.</source> <volume>12</volume>, <fpage>2602</fpage>. <pub-id pub-id-type="doi">10.3390/rs12162602</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Large-scale structure from motion with semantic constraints of aerial images</article-title>,&#x201d; in <conf-name>Pattern Recognition and Computer Vision: First Chinese Conference, PRCV 2018</conf-name>, <conf-loc>Guangzhou, China</conf-loc>, <conf-date>November 23-26, 2018</conf-date> (<publisher-name>Springer International Publishing</publisher-name>), <fpage>347</fpage>&#x2013;<lpage>359</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Goshtasby</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>On the Canny edge detector</article-title>. <source>Pattern Recognit.</source> <volume>34</volume>, <fpage>721</fpage>&#x2013;<lpage>725</lpage>. <pub-id pub-id-type="doi">10.1016/s0031-3203(00)00023-6</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>New grassland riparian zone delineation method for calculating ecological water demand to guide management goals</article-title>. <source>River Res. Appl.</source> <volume>36</volume>, <fpage>1838</fpage>&#x2013;<lpage>1851</lpage>. <pub-id pub-id-type="doi">10.1002/rra.3707</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Foody</surname>
<given-names>G. M.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Status of land cover classification accuracy assessment</article-title>. <source>Remote Sens. Environ.</source> <volume>80</volume>, <fpage>185</fpage>&#x2013;<lpage>201</lpage>. <pub-id pub-id-type="doi">10.1016/s0034-4257(01)00295-4</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Verrelst</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Remote sensing algorithms for estimation of fractional vegetation cover using pure vegetation index values: a review</article-title>. <source>ISPRS J. Photogrammetry Remote Sens.</source> <volume>159</volume>, <fpage>364</fpage>&#x2013;<lpage>377</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2019.11.018</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamylton</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Morris</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Carvalho</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Roder</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Barlow</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mills</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Evaluating techniques for mapping island vegetation from unmanned aerial vehicle (UAV) images: pixel classification, visual interpretation and machine learning approaches</article-title>. <source>Int. J. Appl. Earth Observation Geoinformation</source> <volume>89</volume>, <fpage>102085</fpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2020.102085</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kornilov</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Safonov</surname>
<given-names>I. V.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>An overview of watershed algorithm implementations in open source libraries</article-title>. <source>J. Imaging</source> <volume>4</volume>, <fpage>123</fpage>. <pub-id pub-id-type="doi">10.3390/jimaging4100123</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kutz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cook</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Linderman</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Object based classification of a riparian environment using ultra-high resolution imagery, hierarchical landcover structures, and image texture</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>11291</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-14757-y</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laslier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hubert-Moy</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Corpetti</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dufour</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Monitoring the colonization of alluvial deposits using multitemporal UAV RGB-imagery</article-title>. <source>Appl. Veg. Sci.</source> <volume>22</volume>, <fpage>561</fpage>&#x2013;<lpage>572</lpage>. <pub-id pub-id-type="doi">10.1111/avsc.12455</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jezek</surname>
<given-names>K. C.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Automated extraction of coastline from satellite imagery by integrating Canny edge detection and locally adaptive thresholding methods</article-title>. <source>Int. J. Remote Sens.</source> <volume>25</volume>, <fpage>937</fpage>&#x2013;<lpage>958</lpage>. <pub-id pub-id-type="doi">10.1080/0143116031000139890</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mcllhagga</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>The Canny edge detector revisited</article-title>. <source>Int. J. Comput. Vis.</source> <volume>91</volume>, <fpage>251</fpage>&#x2013;<lpage>261</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-010-0392-0</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Neto</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Verification of color vegetation indices for automated crop imaging applications</article-title>. <source>Comput. Electron. Agric.</source> <volume>63</volume>, <fpage>282</fpage>&#x2013;<lpage>293</lpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2008.03.009</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>M&#xfc;llerov&#xe1;</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gago</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bu&#x10d;as</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Company</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Estrany</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fortesa</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Characterizing vegetation complexity with unmanned aerial systems (UAS) &#x2013; a framework and synthesis</article-title>. <source>Ecol. Indic.</source> <volume>131</volume>, <fpage>108156</fpage>. <pub-id pub-id-type="doi">10.1016/j.ecolind.2021.108156</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Onishi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lse</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Explainable identification and mapping of trees using UAV RGB image and deep learning</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>903</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-79653-9</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Otsu</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>1979</year>). <article-title>A threshold selection method from gray-level histograms</article-title>. <source>IEEE Trans. Syst. Man, Cybern.</source> <volume>9</volume>, <fpage>62</fpage>&#x2013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1109/tsmc.1979.4310076</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rusn&#xe1;k</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Goga</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Michaleje</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>&#x160;ulc Michalkov&#xe1;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>M&#xe1;&#x10d;ka</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Bertalan</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Remote sensing of riparian ecosystems</article-title>. <source>Remote Sens.</source> <volume>14</volume>, <fpage>2645</fpage>. <pub-id pub-id-type="doi">10.3390/rs14112645</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Verhoeven</surname>
<given-names>J. T.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Wetlands in Europe: perspectives for restoration of a lost paradise</article-title>. <source>Ecol. Eng.</source> <volume>66</volume>, <fpage>6</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/j.ecoleng.2013.03.006</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Extraction of vegetation information from visible unmanned aerial vehicle images</article-title>. <source>Trans. Agric. Eng.</source> <volume>31</volume>, <fpage>152</fpage>&#x2013;<lpage>159</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1002-6819.2015.05.022</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Vegetation information extraction in karst area based on UAV remote sensing in visible light band</article-title>. <source>Optik - Int. J. Light Electron Opt.</source> <volume>272</volume>, <fpage>170355</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijleo.2022.170355</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Mapping of green tide using true color aerial photographs taken from a unmanned aerial vehicle. Remote Sensing and Modeling of Ecosystems for Sustainability XIV</article-title>. <source>SPIE</source> <volume>10405</volume>, <fpage>152</fpage>&#x2013;<lpage>157</lpage>. <pub-id pub-id-type="doi">10.1117/12.2271724</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>River delineation from remotely sensed imagery using a multi-scale classification approach</article-title>. <source>IEEE J. Sel. Top. Appl. Earth Observations Remote Sens.</source> <volume>7</volume>, <fpage>4726</fpage>&#x2013;<lpage>4737</lpage>. <pub-id pub-id-type="doi">10.1109/jstars.2014.2309707</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>New research methods for vegetation information extraction based on visible light remote sensing images from an unmanned aerial vehicle (UAV)</article-title>. <source>Int. J. Appl. Earth Observation Geoinformation</source> <volume>78</volume>, <fpage>215</fpage>&#x2013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1016/j.jag.2019.01.001</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Green vegetation extraction based on visible light image of UAV</article-title>. <source>Chin. Environ. Sci.</source> <volume>41</volume>, <fpage>2380</fpage>&#x2013;<lpage>2390</lpage>. <pub-id pub-id-type="doi">10.19674/j.cnki.issn1000-6923.2021.0252</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>