<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2023.1120189</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>3D segmentation of plant root systems using spatial pyramid pooling and locally adaptive field-of-view inference</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Alle</surname>
<given-names>Jonas</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2225260"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gruber</surname>
<given-names>Roland</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2217961"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>W&#xf6;rlein</surname>
<given-names>Norbert</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Uhlmann</surname>
<given-names>Norman</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Clau&#xdf;en</surname>
<given-names>Joelle</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1228662"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wittenberg</surname>
<given-names>Thomas</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/39388"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Gerth</surname>
<given-names>Stefan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/937934"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Fraunhofer Institut f&#xfc;r Integrierte Schaltungen (IIS), Fraunhofer Institute for Integrated Circuits Institut f&#xfc;r Integrierte Schaltungen (IIS), Division Development Center X-Ray Technology</institution>, <addr-line>F&#xfc;rth</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Friedrich-Alexander-Universit&#xe4;t Erlangen-N&#xfc;rnberg, Chair for Visual Computing</institution>, <addr-line>Erlangen</addr-line>, <country>Germany</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Fraunhofer Institut f&#xfc;r Integrierte Schaltungen (IIS), Fraunhofer Institute for Integrated Circuits Institut f&#xfc;r Integrierte Schaltungen (IIS), Division Smart Sensors and Electronics</institution>, <addr-line>Erlangen</addr-line>, <country>Germany</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Manya Afonso, Wageningen University and Research, Netherlands</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Michael Pound, University of Nottingham, United Kingdom; Pushkar Gole, University of Delhi, India</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Stefan Gerth, <email xlink:href="mailto:Stefan.Gerth@iis.fraunhofer.de">Stefan.Gerth@iis.fraunhofer.de</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work and share first authorship</p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Technical Advances in Plant Science, a section of the journal Frontiers in Plant Science</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>04</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1120189</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>03</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Alle, Gruber, W&#xf6;rlein, Uhlmann, Clau&#xdf;en, Wittenberg and Gerth</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Alle, Gruber, W&#xf6;rlein, Uhlmann, Clau&#xdf;en, Wittenberg and Gerth</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>The non-invasive 3D-imaging and successive 3D-segmentation of plant root systems has gained interest within fundamental plant research and selectively breeding resilient crops. Currently the state of the art consists of computed tomography (CT) scans and reconstruction followed by an adequate 3D-segmentation process.</p>
</sec>
<sec>
<title>Challenge</title>
<p>Generating an exact 3D-segmentation of the roots becomes challenging due to inhomogeneous soil composition, as well as high scale variance in the root structures themselves.</p>
</sec>
<sec>
<title>Approach</title>
<p>(1) We address the challenge by combining deep convolutional neural networks (DCNNs) with a weakly supervised learning paradigm. Furthermore, (2) we apply a spatial pyramid pooling (SPP) layer to cope with the scale variance of roots. (3) We generate a fine-tuned training data set with a specialized sub-labeling technique. (4) Finally, to yield fast and high-quality segmentations, we propose a specialized iterative inference algorithm, which locally adapts the field of view (FoV) for the network.</p>
</sec>
<sec>
<title>Experiments</title>
<p>We compare our segmentation results against an analytical reference algorithm for root segmentation (<italic>RootForce</italic>) on a set of roots from Cassava plants and show qualitatively that an increased amount of root voxels and root branches can be segmented.</p>
</sec>
<sec>
<title>Results</title>
<p>Our findings show that with the proposed DCNN approach combined with the dynamic inference, much more, and especially fine, root structures can be detected than with a classical analytical reference method.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>We show that the application of the proposed DCNN approach leads to better and more robust root segmentation, especially for very small and thin roots.</p>
</sec>
</abstract>
<kwd-group>
<kwd>root phenotyping</kwd>
<kwd>root system analysis</kwd>
<kwd>computed tomography</kwd>
<kwd>weakly supervised learning</kwd>
<kwd>sub-labels</kwd>
<kwd>scale invariance</kwd>
<kwd>flood-filling</kwd>
<kwd>convolutional neural networks</kwd>
</kwd-group>
<contract-num rid="cn001">20-3410-2-9-8</contract-num>
<contract-num rid="cn002">INV-008053</contract-num>
<contract-sponsor id="cn001">ADA Lovelace Center for Analytics, Data, Applications<named-content content-type="fundref-id">10.13039/501100018808</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Bill and Melinda Gates Foundation<named-content content-type="fundref-id">10.13039/100000865</named-content>
</contract-sponsor>
<counts>
<fig-count count="9"/>
<table-count count="2"/>
<equation-count count="1"/>
<ref-count count="36"/>
<page-count count="18"/>
<word-count count="13414"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>To optimize and selectively breed resilient and sustainable plant species, automatic or interactive segmentation subterranean root structures from reconstructed CT scans is of great importance for phenotyping plants (<xref ref-type="bibr" rid="B25">Mooney et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B2">Atkinson et&#xa0;al., 2019</xref>). Thus, the quantification of the root-system architecture reaction on biotic and abiotic stress factors is possible. Up to now, most optimization and breeding efforts have relied on phenotyping of above ground structures and completely ignored &#x2018;below-the-ground&#x2019; structures.</p>
<p>In contrast to many conventional methods, such as soil-coring, root washing or <italic>shovelomics</italic> (<xref ref-type="bibr" rid="B34">Trachsel et&#xa0;al., 2011</xref>), the non-invasive scanning and reconstruction of CT-volumes of roots and their successive 3D segmentation is a crucial and powerful approach in this field of research [e.g. <xref ref-type="bibr" rid="B6">De Smet et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B25">Mooney et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B24">Metzner et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B1">Ahmed et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B2">Atkinson et&#xa0;al., 2019</xref>). However, these new possibilities for the non-destructive monitoring of root systems demand high quality data sets of segmented data for quantitative data analysis. This is important since a classical ground truth <italic>via</italic> root excavation and washing, results in a loss of fine root structure and interconnectivity of the whole root system. To this end, the 3D segmentation of the root structures was mainly performed by analytical algorithms based on classical image processing and image analysis methods [e.g., <xref ref-type="bibr" rid="B23">Mairhofer et&#xa0;al., 2011</xref>; <xref ref-type="bibr" rid="B9">Flavel et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B22">Mairhofer et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B8">Flavel et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B11">Gao et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B31">Soltaninejad et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B12">Gerth et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B26">Phalempin et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B7">Ferreira et al., 2022</xref>; <xref ref-type="bibr" rid="B21">Lucas &amp; Vetterlein, 2022</xref>), which, however, are not able to detect roots on all scales equally. See <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> for an example which contains challenging small and fine roots.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Challenges for small root segmentation: <bold>(A)</bold> one vertical 2D slice <italic>S<sub>i</sub>
</italic> from volume <italic>V</italic>
<sub>5</sub> (see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). Yellow box: stem of the cassava plant; red region: planting pot; dark blue region: air; green box: example of soil mixture; magenta and orange marks: two branches of a very thin root. <bold>(B)</bold> Enlarged version of the green box in <bold>(A)</bold> with marked root branches. <bold>(C)</bold> Adjacent slice <italic>S<sub>i+</sub>
</italic>
<sub>2</sub> where the lower root branch (orange) passes a small stone (cyan) with a similar appearance.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g001.tif"/>
</fig>
<p>To overcome the limitations of the current analytical methods, we propose an improved deep neural network based approach, which is (a) able to adapt to different feature scales of the roots to be delineated, (b) can be trained with noisy information, and (c) furthermore is able to efficiently handle large scale 3D data sets of roots during inference. We make use of a classical deep convolutional neural network (DCNN) architecture which is enhanced by a &#x2018;Spatial Pyramid Pooling&#x2019; (SPP) layer (<xref ref-type="bibr" rid="B14">He et&#xa0;al., 2014</xref>) (see Section 4) to gain scale invariance for root segmentation by supporting arbitrarily sized &#x2018;<italic>field of views</italic>&#x2019; (FoVs) (see Section 3.4) as input sample and is furthermore using a weakly supervised training scheme (<xref ref-type="bibr" rid="B17">Khoreva et&#xa0;al., 2017</xref>) (see Section 5).</p>
<p>The work is structured as follows: Section 2 gives an overview of related work, while Section 3 introduces the 3D volumetric CT data of plant root-structures used for training and testing of the modified network. Section 3 also includes information about the labeling of the needed ground truth from an analytical approach, motivates the usage of size varied FoVs, and explains the proposed sub-labeling technique yielding a fine-tuned reference data set.</p>
<p>Section 4 introduces the proposed architecture of the deep neural network including the spatial pyramid pooling layer. In Section 5 the novel training approach is presented including a description of the weakly supervised training loop.</p>
<p>The novel inference algorithm is explained with a detailed walk-through of its functionality in Section 6. Experiments and results are presented in Section 7. Finally, a critical discussion is given in Section 8 and the work is concluded in Section 9.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<p>The challenging task at hand consist of a semantic segmentation of each voxel within a data volume (acquired from reconstructed CT scans) into the two labels <italic>L</italic> &#x2208; {<italic>&#x2018;root&#x2019;, &#x2018;non-root&#x2019;</italic>} denoting either <italic>&#x2018;root&#x2019;</italic> being foreground or <italic>&#x2018;non-root&#x2019;</italic> being background voxels denoting surrounding soil, air, or the planting pot (cf. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). The underlying challenge of this binarization relates to the strong variance of the root diameters, which range from the storage root diameter of &#xd8; = 10 &#x2013; 30&#xa0;mm, to lateral root extensions with diameters in the range of &#xd8; = 0.2&#xa0;mm, as well as varying scanning conditions.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Typical examples of different <italic>cassava</italic> root systems, 3D-segmented with <italic>RootForce</italic> (<xref ref-type="bibr" rid="B12">Gerth et&#xa0;al., 2021</xref>).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" colspan="4" align="center">Training</th>
<th valign="top" align="center">Validation</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-i001.tif"/>
</td>
<td valign="top" align="left">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-i002.tif"/>
</td>
<td valign="top" align="left">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-i003.tif"/>
</td>
<td valign="top" align="left">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-i004.tif"/>
</td>
<td valign="top" align="left">
<inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-i005.tif"/>
</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>V</italic>
<sub>1</sub>: 4,264,632 voxels</td>
<td valign="top" align="left">
<italic>V</italic>
<sub>2</sub>: 10,462,217 voxels</td>
<td valign="top" align="left">
<italic>V</italic>
<sub>3</sub>: 1,596,662 voxels</td>
<td valign="top" align="left">
<italic>V</italic>
<sub>4</sub>: 3,577,330 voxels</td>
<td valign="top" align="left">
<italic>V</italic>
<sub>5</sub>: 2,834,479 voxels</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In the past years various methods and algorithms addressing this task have been proposed. See (<xref ref-type="bibr" rid="B35">Xu et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B19">Li et&#xa0;al., 2022</xref>) for an excellent broad and deep overview of the research area. These can be clustered in two groups, namely (a) traditional knowledge-driven top-down image-processing and analysis methods, and (b) more recently developed data-driven bottom-up approaches employing deep neural network architectures and a large corpus of labeled training data.</p>
<p>Examples of the first category include the &#x2018;<italic>RootForce</italic>&#x2019; algorithm (<xref ref-type="bibr" rid="B12">Gerth et&#xa0;al., 2021</xref>), which solves this task by analyzing the curvature of the gray value profile of the root voxels using classical image processing and analysis approaches. Using the specific mean gray values of the two object types &#x2018;<italic>root</italic>&#x2019; and &#x2018;<italic>non-root</italic>&#x2019;, <italic>RootForce</italic> sorts out all objects which are not in a specific gray value range and thereby generates a mask to indicate possible root voxels. Based on this mask, Frangi&#x2019;s vesselness approach (<xref ref-type="bibr" rid="B10">Frangi et&#xa0;al., 1998</xref>) is applied to create two root volumes for small and large roots, respectively.</p>
<p>Another analytical approach, referred to as <italic>&#x2018;Rootine&#x2019;</italic>, also building upon Frangi&#x2019;s vesselness filter was presented by <xref ref-type="bibr" rid="B11">Gao et&#xa0;al. (2019)</xref> and <xref ref-type="bibr" rid="B26">Phalempin et&#xa0;al. (2021)</xref>, and uses 3D hysteresis thresholding to binarize the resulting image. A similar approach employing region and volume growing method has been proposed by <xref ref-type="bibr" rid="B9">Flavel et&#xa0;al. (2012)</xref> and later published under the name &#x2018;<italic>Root1</italic>&#x2019; (<xref ref-type="bibr" rid="B8">Flavel et&#xa0;al., 2017</xref>).</p>
<p>Further analytical approaches have been suggested by <xref ref-type="bibr" rid="B23">Mairhofer et&#xa0;al. (2011)</xref>, who have developed the so-called <italic>&#x2018;RooTrak&#x2019;</italic> algorithm. This approach makes use of slice wise level-set techniques combined with the so-called <italic>Jensen-Shannon</italic> divergence to segment plant roots in volumetric data sets. Later, this procedure was expanded to enable the tracking of multiple touching roots using the iterative closest point (ICP) algorithm to match and separate multiple possible roots of different root systems between slices (<xref ref-type="bibr" rid="B22">Mairhofer et&#xa0;al., 2015</xref>).</p>
<p>The second group of methods relate to deep-learning inspired approaches, which have recently been introduced to this field of research, e.g., a multi-resolution encoder-decoder network to handle the diverse scales of root systems was proposed by <xref ref-type="bibr" rid="B31">Soltaninejad et&#xa0;al. (2020)</xref>. They propose a parallel processing pipeline featuring a path of high resolution with a small receptive field and a path of lower resolution but large receptive field to segment 3D volume patches. This approach has been built on top of the well-known U-Net architecture (<xref ref-type="bibr" rid="B27">Ronneberger et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B4">Cicek et&#xa0;al, 2016</xref>) used for semantic segmentation.</p>
<p>
<xref ref-type="bibr" rid="B30">Smith et&#xa0;al. (2020)</xref> also employ a U-Net to automatically segment <italic>rhizotrons</italic> in RGB photographs of <italic>Chicory</italic> root systems. Here, the network was trained and validated on a set of 50 manually labeled 2D images and compared to a baseline using Frangi&#x2019;s vesselness filter (<xref ref-type="bibr" rid="B10">Frangi et&#xa0;al., 1998</xref>). Using 3D-MRI data, the work by <xref ref-type="bibr" rid="B36">Zhao et&#xa0;al. (2020)</xref> also utilized a U-Net architecture for the segmentation of roots using custom loss modifications to reduce erroneous disconnected roots.</p>
<p>Alternatively, <xref ref-type="bibr" rid="B29">Shen et&#xa0;al. (2020)</xref> make use of a slightly modified <italic>DeepLabv3+</italic> network architecture (<xref ref-type="bibr" rid="B3">Ayhan, 2020</xref>) to segment RGB photographs of cotton root systems. Specifically, they introduced a sub-pixel convolution to the up-sampling path of the model and compared their results with manually annotated data as well with the results of a U-Net based segmentation. <xref ref-type="bibr" rid="B33">Thesma and Mohammadpour Velni (2023)</xref> propose a conditional generative adversarial network (cGAN) to augment their available root data set of RGB photographs from plant roots grown in clear gel. They further used a SegNet trained on both synthetic and real data for segmentation of fine root systems.</p>
<p>Our own proposed network (see <bold>Section 4</bold>) builds upon the work by <xref ref-type="bibr" rid="B14">He et&#xa0;al. (2014)</xref>, who introduced the so-called &#x2018;Spatial Pyramid Pooling network&#x2019; (&#x2018;SPP-network&#x2019;) with the goal to alleviate the fixed size constraint of input images for classical deep convolutional neural Networks (DCNNs) (<xref ref-type="bibr" rid="B28">Schmidhuber, 2015</xref>; <xref ref-type="bibr" rid="B18">Krizhevsky et&#xa0;al., 2017</xref>). After feature extraction with several convolution layers and by introducing a pooling method, that generates fixed size outputs from arbitrary sized inputs a pyramid of feature maps is generated, ranging from a coarse-to-fine detail level. This allows training a network with input data of varying image sizes, resulting in a more robust classification as output with respect to scale variance. Specifically, for this contribution we employed this SPP-net (see <bold>Section 4</bold>) and extended it to the domain of CT-data and 3D root segmentation.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Data</title>
<p>The data set used for this project was generated from fourteen high-resolution CT root scans from different potted <italic>Cassava</italic> plants with the same set of scanning and calibration parameters. From the projective raw data, cuboid volumes with spatial dimensions of 17.9&#xa0;cm &#xd7; 17.9&#xa0;cm &#xd7; 15.7&#xa0;cm were reconstructed, and with a voxel edge length of 175 &#x3bc;m resulted in data volumes of 1024 &#xd7; 1024 &#xd7; 900 voxels per plant. For more details about the scanning see <xref ref-type="bibr" rid="B12">Gerth et&#xa0;al. (2021)</xref>
</p>
<sec id="s3_1">
<label>3.1</label>
<title>Data description</title>
<p>
<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> provides typical examples from the fourteen scanned and 3D-segmented root systems of cassava plants. The bottom row lists the count of detected root voxels within each volume. The 3D-segmentations of the depicted root structures have been generated by the <italic>RootForce</italic> algorithm (<xref ref-type="bibr" rid="B12">Gerth et&#xa0;al., 2021</xref>) with a vesselness filter in the range of 0.2&#x2013;0.5 mm resulting in a volume of small roots, and furthermore used a 3D-Gauss filter (&#x3c3; = 0.5mm) to support the extraction of large roots (see <bold>Section 2</bold>). Both obtained volumes are binarized and merged yielding the final binary root segmentation.</p>
<p>
<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref> shows one vertical 2D image slice <italic>S<sub>i</sub>
</italic> extracted from volume <italic>V</italic>
<sub>5</sub>. This side view depicts the large (bright) stem of the cassava plant (denoted by the yellow box in the top center) surrounded by the (dark) soil mixture, the planting pot below (red region) and the air above (dark blue region). The soil mixture is rather inhomogeneous with many components of differing gray values, including small pebbles. To enhance visibility of the very thin root structures depicted in the soil, the contrast was increased. One challenge in this exemplary data is highlighted in the green boxes in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref> and enlarged <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>, where two branches of a very thin root (orange and magenta) are denoted. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref> depicts the two branches of the same root in the adjacent image slice <italic>S<sub>i+</sub>
</italic>
<sub>2</sub>, (shifted two voxels perpendicular to the slicing plane) passing a small stone (cyan) with very a similar roundish structure and gray level values.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Data split</title>
<p>For the training, validating, and testing of the proposed SPP-network (see <bold>Section 4</bold>) the fourteen CT scans were divided. Four volumes (<italic>V</italic>
<sub>1</sub> &#x2013; <italic>V</italic>
<sub>4</sub>) were used as training data and one volume (<italic>V</italic>
<sub>5</sub>) for the validation of the deep neural network. The remaining nine (<italic>V</italic>
<sub>6</sub> &#x2013; <italic>V</italic>
<sub>14</sub>) scans were used for testing. The training volumes were chosen carefully with the distinct aim of providing a diverse set of root examples despite the otherwise identical physical parameters.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Labeled training data</title>
<p>When training a DCNN to delineate the complex root structures into the fore- and background labels <italic>&#x2018;root&#x2019;</italic> and <italic>&#x2018;non-root&#x2019;</italic>, a sufficient amount of adequately labeled ground truth data is required. However, there is only limited availability of training data with a satisfactory image and label quality, thus hindering the training process. Since manual annotation of such CT data is time consuming and error prone, segmentation results of the training data <italic>D</italic> = <italic>{V</italic>
<sub>1</sub>, <italic>V</italic>
<sub>2</sub>, <italic>V</italic>
<sub>3</sub>, <italic>V</italic>
<sub>4</sub>
<italic>}</italic> (see <bold>Section 3.2</bold>) obtained from the <italic>RootForce</italic> algorithm were used as label volumes <italic>L</italic>
<sub>0</sub> (see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>) in an <italic>initial</italic> training data set <italic>T</italic>
<sub>0</sub> = (<italic>D</italic>, <italic>L</italic>
<sub>0</sub>). The deep neural network (see <bold>Section 4</bold>) was trained by a weakly supervised learning approach, as proposed by <xref ref-type="bibr" rid="B17">Khoreva et&#xa0;al. (2017)</xref>, to recursively refine the data set. During the weakly supervised learning iterations, the segmentation results of the network <italic>N</italic>
<sub>0</sub> from the first training iteration are used as label volumes for subsequent training iterations, see <bold>Section 5</bold> for details.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Sample fields of view</title>
<p>Due to memory restrictions, an entire volumetric 3D scan of the roots (with a total of 1024 &#xd7; 1024 &#xd7; 900 voxels) cannot be processed at once. Therefore, we consider small sub-volumes or small 3D-patches from the complete volumes as input samples. Even though only the center voxel of each patch is classified into either label &#x2018;<italic>root</italic>&#x2019; or &#x2018;<italic>non-root</italic>&#x2019;, the surrounding volume patch provides important contextual information. Consequently, the reference label of one sample is <italic>not</italic> a classification of the whole volume patch, but a binary label for the center voxel, encoding either &#x2018;<italic>root</italic>&#x2019; or &#x2018;<italic>non-root</italic>&#x2019;.</p>
<p>In order to detect thin <italic>and</italic> thick roots (see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>; <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>) with equal accuracy, similar structural root features need to be represented across scales. To restrict the introduction of additional noise and expensive preprocessing during inference, scaling techniques utilizing filtering were avoided. Instead, as all samples originate from image data with the same voxel resolution, we vary the edge length of the patches for each sample; hence, the training sample size of an input patch varies only in the voxel count and thus the dimensions of the cuboid, but not in the physical voxel size. In the remains of this work, we will refer to the varying sample size as &#x2018;<italic>field of view&#x2019;</italic> (FoV) which is created by extracting the neighboring region around each voxel. This region can extend between <italic>r</italic> = 2 to <italic>r</italic> = 7 voxels in each direction, hence resulting in cubic samples with edge lengths in the range from <italic>l</italic> = 5 to <italic>l</italic> = 15 voxels.</p>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Sub-labeling</title>
<p>Since every volumetric data sample corresponds to a distinct voxel, which is either of class &#x2018;<italic>root&#x2019;</italic> or class &#x2018;<italic>non-root</italic>&#x2019;, a naively constructed training data set would inherit a skewed class distribution, as there are much less root voxels than non-root voxels present in each CT scan. Randomly drawing samples from the pool of possible non-root voxels to match the count of root voxels, could yield many trivial negative samples, such as volume patches depicting only soil or air around the plant pot, see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>. To provide an adequate training data set with an ideally uniform class distribution, the applied data preprocessing includes a categorization of the training and validation samples into sub-labels, which distinguish the center voxel of the volume patch from the surrounding voxels.</p>
<p>Because during the later inference, the proposed network (see Section 4) shall be provided a volume patch, and its output should predict one of the classes &#x2018;<italic>root</italic>&#x2019; or &#x2018;<italic>non-root</italic>&#x2019; for <italic>only</italic> its center voxel, it is important to train the network with a sufficient set of &#x2018;<italic>difficult negative</italic>&#x2019; examples. Specifically, in such &#x2018;difficult negative&#x2019; samples, some root voxels should be present in the volume patch, while the center voxel <italic>is not</italic> of the class &#x2018;<italic>root</italic>&#x2019;. In this way, the network will be prevented from classifying the volume patch as a whole and instead is taught to differentiate between the contents in the center voxel and its surrounding.</p>
<p>Furthermore, another challenge with respect to the application of deep convolutional neural networks to plant root segmentation is the misclassification of voxels depicting the plant pot as class &#x2018;<italic>root</italic>&#x2019;. As the non-trivial geometry of the pot base (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>) could not be easily extracted or excluded by a mask, the sub-labels also differentiate within the <italic>&#x2018;non-root&#x2019;</italic> class, namely between <italic>&#x2018;sediment&#x2019;</italic> and <italic>&#x2018;else&#x2019;</italic> contents, where &#x2018;<italic>else</italic>&#x2019; is used as collective term for all voxels depicting the plant pot or air. This approach allows further specification for the included samples of the negative &#x2018;<italic>non-root</italic>&#x2019; class in the training and validation data. For example, &#x2018;<italic>non-root</italic>&#x2019; samples, which contain roots in the surrounding but not in the center voxel and are located next to the edges of the plant pot.</p>
<p>As depicted in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, for the categorization of the training samples into adequate sub-labels, two label volumes <italic>V</italic>
<sub>Roots</sub> and <italic>V</italic>
<sub>Sediment</sub> are employed (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>): The first volume, <italic>V</italic>
<sub>Roots</sub>, is a binary volume and depicts an approximate labeling <italic>V</italic>
<sub>Roots</sub> = <italic>L<sub>i</sub>
</italic> of the &#x2018;<italic>root</italic>&#x2019;/&#x2019;<italic>non-root</italic>&#x2019; classification, which in the first approximation is obtained from a <italic>RootForce</italic> segmentation (see Section 3.3). The second label volume <italic>V</italic>
<sub>Sediment</sub> approximates a sediment segmentation, which is generated from the inverse <italic>V</italic>
<sub>Roots</sub> label volume. Additional image processing steps are applied in order to remove air pockets in the soil and the plant pot from the sediment labeling. However, a conservative box-mask, which also excludes several sediment voxels, was needed for the lower volume due to the complex geometry of the pot bottom. Hence, it is assured that no voxels from &#x2018;pot&#x2019; or &#x2018;air&#x2019; are contained in V<sub>Sediment</sub>, and thereby the sediment label volume helps differentiating between the content classes &#x2018;sediment&#x2019; and &#x2018;else&#x2019;. The original gray level CT-volume, depicted in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>, only serves for a better description of the process, as its information is not needed to generate the sub-labels.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>2D description of the volume sample categorization into nine sub-labels (<italic>&#x2018;root-root&#x2019;</italic>, <italic>&#x2018;root-sediment&#x2019;</italic>, <italic>&#x2018;root-else&#x2019;</italic>, <italic>&#x2018;sediment-root&#x2019;</italic>, <italic>&#x2018;sediment-sediment&#x2019;</italic>, <italic>&#x2018;sediment-else&#x2019;</italic>, <italic>&#x2018;else-root&#x2019;</italic>, <italic>&#x2018;else-sediment&#x2019;</italic> and <italic>&#x2018;else-else&#x2019;</italic>). <bold>(A)</bold> input data and pre-labeled masks <italic>V</italic>
<sub>Roots</sub> and <italic>V</italic>
<sub>Sediment</sub> corresponding to one scan; <bold>(B)</bold> extraction of individual volume patches according to the current FoV for each interior voxel; <bold>(C)</bold> example of the depicted contents within the volume patches; <bold>(D)</bold> overview of possible sub-label combinations differentiating between combinations of content in the center voxel and the surrounding voxels.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g002.tif"/>
</fig>
<p>The following steps are repeated for all possible radius sizes <italic>r</italic>&#xa0;&#x2208;&#xa0;{2, 3, 4, 5, 6, 7} of the FoV: For every &#x2018;interior voxel&#x2019; of both the approximated ground truth volume <italic>V</italic>
<sub>Roots</sub> as well as the approximated sediment label volume <italic>V</italic>
<sub>Sediment</sub>, sample patches, which are centered on the current voxel and sized according to the current radius <italic>r</italic>, are extracted and then used to categorize each sample. The denotation of &#x2018;interior voxels&#x2019; refers to those voxels, which are at least <italic>d</italic> = &#x230a;<italic>r</italic>/2&#x230b; voxels away from the volume&#x2019;s edges, such that the FoV can be placed around the voxel without needing to pad the volume.</p>
<p>
<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref> depicts the interior voxels of the volume in <italic>white</italic>, while for the FoV the center voxel is marked in <italic>red</italic>, and <italic>yellow</italic> indicates the voxels in the surrounding regions. Exemplary resulting patches are shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2C</bold>
</xref>.</p>
<p>As motivated earlier, the categorization of all volume patches into sub-labels shall differentiate the occurring content combinations of the center voxel and the surrounding voxels, where the content classes are <italic>&#x2018;root&#x2019;</italic>, <italic>&#x2018;sediment&#x2019;</italic>, and <italic>&#x2018;else&#x2019;</italic> &#x2013; surrogating &#x2018;pot&#x2019; and &#x2018;air&#x2019;. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2D</bold>
</xref> shows the considered sub-labels in a table-like structure, where the contents of the center voxel are organized along the rows and the contents of the surrounding voxels are organized along the columns. This is also reflected by the respective color-coding of <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>. The secondary table headers are color-coded to match the three content types (<italic>&#x2018;root&#x2019;</italic>, <italic>&#x2018;sediment&#x2019;</italic> and <italic>&#x2018;else&#x2019;</italic>). In total nine different combinations of sub-labels are considered. The first row contains the sub-labels, which correspond to the original &#x2018;<italic>root</italic>&#x2019; class, while all other sub-labels are of the &#x2018;<italic>non root</italic>&#x2019; class. Each table entry gives an exemplary depiction of the three volume patches (edge colors indicate the affiliations to the source volumes). With the two binary volume patches, <italic>V</italic>
<sub>Roots</sub> and <italic>V</italic>
<sub>Sediment</sub>, and some binary logical operators, every sample can now be categorized into a unique sub-label. The output of this process is a table for each FoV radius size <italic>r</italic> &#x2208; {2, 3, 4, 5, 6, 7}, containing the total amount of available samples per sub-label and storing the respective sample coordinates.</p>
<p>By specifying the sample counts per sub-label, a class-balanced training data set can now be constructed. To this end, we define a distribution, dependent on the difficulty and availability of each sub-label.</p>
<p>This above-described procedure of sub-label generation is repeated for all five CT scans used as training or validation volume. Where possible, similar distributions for every volume and FoV combination were used. However, this goal is not always attainable since the availability of the sub-label depends on the spatial extend of the addressed root system and the FoV radius <italic>r</italic>. Specifically for the later, the proportion of samples containing sub-labels, which encode for a mix of content types, scales proportionally with the FoV size.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Deep neural network architecture</title>
<p>For our experiments we employ a classical deep convolutional neural network (DCNN) structure (<xref ref-type="bibr" rid="B28">Schmidhuber, 2015</xref>; <xref ref-type="bibr" rid="B18">Krizhevsky et&#xa0;al., 2017</xref>) with a fully connected layer at the end as depicted in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Schematic depiction of the employed network architecture. Specifically, the spatial pyramid pooling layer in the center of the network consists itself of multiple internal layers, which are processed in parallel. As the input can be of arbitrary sizes, the input and output dimensions are generically denoted as <italic>h</italic> &#xd7; <italic>w</italic> &#xd7; <italic>d</italic>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g003.tif"/>
</fig>
<p>However, one known downside of this network structure is the lack of sufficient scale invariance. This is a crucial issue as due to their fractal geometry (<xref ref-type="bibr" rid="B5">Dannowski &amp; Block, 2005</xref>) the root structures depict strong similarities across different scales. Hence, the DCNN needs to learn scale invariant features, which can be reused on different scales to detect all types of roots robustly. This necessity is incorporated into the DCNN by including a &#x2018;Spatial Pyramid Pooling&#x2019; (SPP) layer (<xref ref-type="bibr" rid="B14">He et&#xa0;al., 2014</xref>). This SPP-layer enables the network to learn and generalize scale-invariant features, and furthermore, relieves the size constraint of the input samples, which is inflicted by the fully connected layers. In order to utilize both advantages, different sized volume patches &#x2013; also referred to as &#x2018;<italic>fields-of-view</italic>&#x2019; (FoVs) (see <bold>Section 3.4</bold>) &#x2013; which are created in the sub-labelling process (see <bold>Section 3.5</bold>) - are used as input for training of the same network.</p>
<p>The DCNN incorporates a total of five successive convolutional layers with an increasing count of channels to extract characteristic features from the gray level input sample <italic>via</italic> 5 &#xd7; 5 &#xd7; 5 kernels.</p>
<p>The subsequent spatial pyramid pooling (SPP) layer (<xref ref-type="bibr" rid="B14">He et&#xa0;al., 2014</xref>) consists of three parallel adaptive pooling layers, each generating a feature description on a different fixed scale. The sliding kernel of each pooling layer changes its size adaptively to ensure fixed output dimensions. Thereby, the total complexity of the feature description remains constant for arbitrarily sized inputs. For this application a pyramid consisting of the three levels is used, with each level having a cubic output shape with edge lengths of 1, 2, and 4 voxels, respectively.</p>
<p>After flattening and concatenation, the output of the SPP-layer is forwarded into the first layer of a classical Multi-Layer Perceptron (MLP). Three fully connected layers reduce the extracted feature vector to a final prediction of the two classes <italic>&#x2018;root&#x2019;</italic> and &#x2018;<italic>non-root</italic>&#x2019;. The last layer of the DCNN uses the <italic>softmax</italic> activation function to scale the output value to the probability of belonging to the respective class.</p>
<p>For regularization purposes, dropout (<xref ref-type="bibr" rid="B32">Srivastava et&#xa0;al., 2014</xref>) is used with a probability of 40% after the convolution layers and 70%&#xa0;after the fully connected layers. ReLU is employed as activation function.</p>
</sec>
<sec id="s5">
<label>5</label>
<title>Network training</title>
<p>Within the scope of this work, a so-called <italic>weakly supervised training</italic> scheme as proposed by Khoreva et&#xa0;al. [<sup>26</sup>], is employed by using the available CT training data (Section 3.1) with <italic>iteratively</italic> improving label data from <italic>RootForce</italic> (Section 3.3). In each iteration, the sub-labeling (Section 3.5) is used to balance the data set, and a dynamic inference algorithm (Section 6) generates the improved labeling.</p>
<p>Within the <italic>weakly supervised learning</italic> approach, a network model <italic>N<sub>i</sub>
</italic> is trained on an iteratively improving training data set <italic>T<sub>i</sub> = (D, L<sub>i</sub>)</italic>. It has been observed by Khoreva et&#xa0;al. <bold>[</bold>
<sup>26</sup>
<bold>]</bold>, that when re-applying the network model <italic>N</italic>
<sub>i</sub> on the training data D, the output <italic>N<sub>i</sub>
</italic> (<italic>D</italic>) <italic>= O</italic>
<sub>i</sub> of the network captures the shapes of the objects (in this case the root structure) significantly better than the label data during the training. This has inspired a recursive training procedure, where the achieved output labels replace the training labels <italic>L<sub>i+1</sub> </italic>= <italic>O<sub>i</sub>
</italic> and together with the original training data <italic>D</italic> serve as a new training data set <italic>T<sub>i</sub>
</italic>
<sub>+1</sub> = (<italic>D</italic>, <italic>L<sub>i</sub>
</italic>
<sub>+1</sub>) = (<italic>D</italic>, <italic>N<sub>i</sub>
</italic> (<italic>D</italic>)) for another round of training.</p>
<p>Due to the generalizing capabilities of deep neural networks, incorrect outlier-labels become more aligned to the majority of correct labels over the course of recursively re-evaluating the training data with the neural network. Hence, a robustly trained neural network will smooth out possible label noise and thus yield an improved labeled training data set <italic>T<sub>i</sub>
</italic>
<sub>+1</sub> = (<italic>D</italic>, <italic>L<sub>i+</sub>
</italic>
<sub>1</sub>) = (<italic>D</italic>, <italic>N<sub>i</sub>
</italic> (<italic>D</italic>)). However, the weakly supervised learning paradigm cannot guarantee an improvement. Mistakes in the label data will not be removed generally, but depending on the training scheme and data set, label noise can be reduced by smoothing of the data-label correspondences. An amplification of mistakes in the training data set, resulting in &#x2018;<italic>label drift</italic>&#x2019;, was not observed for this work. While difficult to verify, the fine-tuning of the training data set <italic>via</italic> sub-labeling (see Section 3.5) might steer the training process towards greater robustness and thereby avoid label drift. The recursive training-evaluation cycle was repeated three times, each starting with a new network model <italic>N<sub>i</sub>
</italic> from scratch.</p>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> summarizes the training iterations of the weakly supervised learning scheme of the proposed SPP-network (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4E</bold>
</xref>). As mentioned, the first iteration <italic>i = 0</italic> uses the segmentation calculated by the <italic>RootForce</italic> (<xref ref-type="bibr" rid="B12">Gerth et&#xa0;al, 2021</xref>) algorithm as target labels <italic>L</italic>
<sub>0</sub> (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>). Each iteration applies the sub-label categorization (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4D</bold>
</xref>) from Section 3.5 based on the current approximated target labels <italic>L<sub>i</sub>
</italic> (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref>) and the gray value CT data <italic>D</italic> (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Schematic overview of the weakly supervised training loop. From gray value CT data <italic>D</italic> <bold>(A)</bold>, together with corresponding label data <italic>L<sub>i</sub>
</italic> (provided by the <italic>RootForce</italic> algorithm for <italic>i=0</italic>) <bold>(B)</bold>, the training data <italic>T<sub>i</sub>
</italic> = (<italic>D</italic>, <italic>L<sub>i</sub>
</italic>) <bold>(C)</bold> is used to construct a sub-label balanced data set <bold>(D)</bold> which is then used to train a SPP-network model <italic>N<sub>i</sub>
</italic> <bold>(E)</bold>. The trained SPP-network <italic>N<sub>i</sub>
</italic> is applied on the training data <italic>D</italic> using the proposed dynamic interface <bold>(F)</bold>, yielding a new labeling <italic>L<sub>i+1</sub>
</italic> <bold>(G)</bold>. In the next training iteration, this output is used as new (and improved) reference label <italic>L<sub>i+1</sub>
</italic> <bold>(H)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g004.tif"/>
</fig>
<p>As shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4F</bold>
</xref>, during the inference cycle the training and validation CT volumes were assessed with the custom inference algorithm, see Section 6. According to the work by Khoreva et&#xa0;al. [<sup>26</sup>] as well as our experiments, the resulting root segmentation <italic>O<sub>i</sub>
</italic> shows improvements over the noisy label data <italic>L<sub>i</sub>
</italic>. In the next training iteration, this improved segmentation replaces the previously used target label volume <italic>L<sub>i</sub>
</italic>
<sub>+1</sub> = <italic>O<sub>i</sub>
</italic>, hence <italic>T<sub>i</sub>
</italic>
<sub>+1</sub> = (<italic>D</italic>, <italic>L<sub>i</sub>
</italic>
<sub>+1</sub>) = (<italic>D</italic>, <italic>N<sub>i</sub>
</italic> (<italic>D</italic>)). With each iteration, all sub-steps are repeated, and a new network <italic>N<sub>i+1</sub>
</italic> is trained from scratch. A total of three cycles (i \in {0, 1, 2}) of recursive training were iterated. The resulting network <italic>N</italic>
<sub>2</sub> after the last cycle is fixed as final proposal. In Section 7 and Section 8 the results of this last network model <italic>N</italic>
<sub>2</sub> applied on a completely disjoint test set are presented and discussed.</p>
<p>During the training process, the <italic>cosine annealing learning rate schedule</italic> was applied, as proposed by <xref ref-type="bibr" rid="B20">Loshchilov and Hutter (2016)</xref>. The combination of the learning rate schedule together with <italic>Snapshot Ensembling</italic> (<xref ref-type="bibr" rid="B15">Huang et&#xa0;al., 2017</xref>) was investigated, and showed that a schedule with only one cosine annealing cycle and no ensembling was most promising with respect to the application and the data. Thus, our schedule function had the form</p>
<disp-formula>
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mi>t</mml:mi>
<mml:mi>T</mml:mi>
</mml:mfrac>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>)</mml:mo>
<mml:mo>)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where &#x3b7;<italic>
<sub>t</sub>
</italic> &#x2208; &#x211d;<sup>+</sup> is the learning rate calculated at training step <italic>t</italic> &#x2208; &#x2115;, &#x3b7;<sub>0</sub> &#x2208; &#x211d;<sup>+</sup> is the initial learning rate, and <italic>T</italic> &#x2208; &#x2115; is the total count of training steps. Moreover, <italic>T</italic> corresponds to the total count of training samples <italic>|D|</italic>, divided by the batch size <italic>b</italic> &#x2208; &#x2115; and the update frequency of the schedule <italic>f</italic> &#x2208; &#x2115;, <italic>T</italic> = <italic>|D|/bf</italic>. For all iterations the batch size <italic>b</italic> was fixed to <italic>b</italic> = 256 and the update frequency to <italic>f</italic> = 4,000, where <italic>f</italic> is given in batches, i.e., the step index <italic>t</italic> increments every 4,000 batches. At the end of the schedule the model has seen all training samples once, though, the same voxels are seen multiple times, once for each available FoV size.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Dynamic inference</title>
<p>Since the proposed network architecture (see <bold>Section 4</bold>) only evaluates one voxel (based on its FoV) at a time, the process to segment one complete root volume by iterating through every voxel independently is very time consuming. Additionally, independent voxel evaluations are prone to misclassifications randomly distributed across the volume. Even though such misclassifications can be addressed by adequate post-processing steps, as e.g., by a <italic>connected component analysis</italic> (CCA, see <bold>Section 8.2</bold>), post-processing is usually undesirable and suffers from inflexible thresholds.</p>
<p>Furthermore, the network&#x2019;s architecture with the above introduced SPP-layer was selected with the dedicated goal of inferring root samples of arbitrary input sizes and scales. Leveraging this advantage with a na&#xef;ve voxel-wise sequential inference becomes challenging, as it would require evaluating the complete volume multiple times on different scales. This approach would not only increase the inference time dramatically (up to an order of multiple days) but would also require an adequate process to merge the segmentations from different scales into one joint result, which is a non-trivial challenge.</p>
<p>For these reasons a custom inference approach &#x2013; based on the well-known flood-filling paradigm from classical image processing &#x2013; was developed to not only speed up the inference of one root volume, but also simultaneously improve the segmentation quality. Hence, in an iterative process, only those voxels are chosen for evaluation, which are most likely to contain a root. This selection of possible root voxels is guided by previously evaluated voxels within a certain neighborhood, hence, the independence of the aforementioned voxel-wise inference can be disrupted. Also, this approach eliminates the need for post-processing as it automatically generates a connected root structure. Furthermore, the results are improved by adaptively changing the size of the FoV which is used for the classification of each voxel into the classes <italic>&#x2018;root&#x2019;</italic> or <italic>&#x2018;non-root&#x2019;</italic>.</p>
<p>The workflow of the proposed inference algorithm is shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> and will be discussed step by step in the next sections.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Workflow of the inference algorithm. By iteratively evaluating only neighbor voxels of known roots (&#x2018;growing phase&#x2019;), the algorithm works in a flood-filling manner, by choosing 'yes' at <bold>(A)</bold>. Different sizes for the field of view (FoV) support the trade-off between classification confidence and spatial accuracy. The border of the root is detected adaptively in the &#x2018;pruning phase&#x2019; with the most accurate/smallest FoV, by choosing 'yes' at <bold>(B)</bold>. The inference terminates when all queued voxels have been processed <bold>(C)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g005.tif"/>
</fig>
<sec id="s6_1">
<label>6.1</label>
<title>Flood-filling approach</title>
<p>The iterative process starts at the top of the flow graph in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> by (manually) choosing a set of initial seed voxels, of which the user is certain that at least one of them belongs to the class <italic>&#x2018;root&#x2019;</italic>. These seed voxels are then pushed into a queue, together with the information that at processing time the largest available FoV size should be used for them. Even though the diagram in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> depicts the workflow for a single voxel at a time, multiple voxels can be pulled from the queue at once, to make use of the more efficient batch evaluation. When a voxel is pulled off the queue, its FoV (with radius <italic>r</italic>) from the neighboring voxels is extracted from the volume and prepared as a sample for the network&#x2019;s input layer. If the network classifies the actual center voxel as <italic>&#x2018;root&#x2019;</italic>, the downwards arrow from diamond (a) (in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>) is followed.</p>
<p>In the example depicted in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>, a single seed voxel was chosen. Its queued state is represented by a blue colored disk in the respective voxel. The largest used FoV size in the example has a radius of <italic>r</italic> = 3 voxels and an edge length of <italic>l</italic> = 7 voxels, the related voxels are denoted in blue. <xref ref-type="fig" rid="f6">
<bold>Figure 6E</bold>
</xref> summarizes the color codings of all depictions. Next, in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>, the network has classified the seed voxel as <italic>&#x2018;root&#x2019;</italic> with the given FoV (drawn as a red box with a red circle in the center voxel) and the voxel is thus represented (labeled) with a filled-in colored blue voxel. The process will then queue all yet unseen neighboring voxels (of the current voxel) for further evaluation as it is shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref>. This step is denoted as &#x2018;growing phase&#x2019;, since it fills the queue and allows the detected root structure to grow. While the queue is not empty, and the newly queued voxels are classified to belong to the &#x2018;<italic>root</italic>&#x2019; class, these steps are repeated. This is depicted in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> by looping back to the top in diamond (c), hence iteratively evaluating the collected and yet unseen neighbor voxels in the queue.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Simplified illustrations of iteration steps during the initial <italic>growing phase</italic> of the seed voxel <bold>(A&#x2013;D)</bold>, premature edge detection <bold>(F&#x2013;J)</bold>, tapering root <bold>(K&#x2013;O)</bold> and <italic>pruning phase</italic> <bold>(P-T)</bold>. <bold>(E)</bold> shows a legend of the used color coding. Note that unseen '\emph{root}' and '\emph{non-root}' areas are depicted as a simplification of the real, non-discretized structures.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g006.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6D</bold>
</xref> illustrates a slightly progressed state of the root volume with a growing amount of already classified as well as queued voxels. In contrast to a na&#xef;ve brute-force approach, where <italic>each</italic> voxel in the complete CT-volume is evaluated and classified from top to bottom, this process operates similar to a classical &#x2018;breadth-first&#x2019; flood-filling algorithm, as it starts at a seed voxel and locally expands in every direction equally, until a stopping criterion is met. Furthermore, to avoid evaluating every voxel in the volume, the goal is to stop this growing or &#x2018;flooding&#x2019; process at the boundary of the already detected root structures. However, as these boundaries are not known beforehand, they must be detected as precisely as possible during the expansion process.</p>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>Boundary detection</title>
<p>To robustly detect the boundaries between the root structures and the surrounding soil, advantages are made of the possibility to process FoVs of different spatial sizes. It can be assumed that large FoVs will lead to predictions with a high degree of certainty, where the network is confident in its classification, as it has much contextual information on which to base its estimate. Furthermore, it is assumed that small FoVs will lead to predictions which are locally more accurate, since smaller FoVs might detect the exact voxel position of a root-boundary with higher spatial accuracy than a large FoV, as there exist fewer off-center voxels competing for attention of the network. In contrast, a large FoV can contain many root voxels, none of which are in the center position. This effect could erroneously skew the prediction towards the <italic>&#x2018;root&#x2019;</italic> class, especially when those root voxels are located directly next to a center voxel.</p>
<p>Under these considerations, the best achievable boundary detection of the root structures will likely be computed with small FoVs. For this reason, the proposed evaluation process starts with the largest FoV around the seed voxels, since it allows going from the most confident classification (large FoVs) to the spatially most accurate one (small FoVs). Hence, if a voxel is pulled from the queue and is classified as &#x2018;<italic>non-root</italic>&#x2019;, the size of the FoV is reduced to a smaller radius and the same voxel is pushed in the queue again for further evaluation. This extends the formerly mentioned loop in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> by an alternative route, following the &#x2018;no&#x2019;-arrows from diamond (a) and again from diamond (b), if the currently used FoV is not the smallest one possible.</p>
<p>As depicted in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6F</bold>
</xref>, this effect can happen close to the boundary of the root. Here, the currently processed voxel was queued with the largest FoV with radius <italic>r</italic> = 3, and edge size 7. However, in this example the network classifies the voxel as &#x2018;non root&#x2019;, indicated by a colored cross in the figure. With this classification result, the voxel is inserted in the queue again, now with a smaller FoV. Hence, in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6G</bold>
</xref> the next smaller FoV with radius <italic>r</italic> = 2 and edge size 5 is depicted with a green color-coding. The reduced FoV size now helps the network to focus more on the actual center voxel of the FoV. When the voxel is then classified as <italic>&#x2018;root&#x2019;</italic>, the following neighbor voxels will continue to use the smaller FoV, as can be seen in <xref ref-type="fig" rid="f6">
<bold>Figures&#xa0;6H, I</bold>
</xref> respectively. If the voxel is still classified as &#x2018;<italic>non-root</italic>&#x2019;, the FoV size is further reduced as shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6J</bold>
</xref>. Here a neighbor of the previously discussed voxel was classified as &#x2018;<italic>non-root</italic>&#x2019; with a FoV of edge size 5. In this way, the voxels of the boundary layers of the root structures are grown by smaller FoVs. This especially comes into effect for regions in which the gradient of the gray-level values promotes a premature edge detection for larger FoVs.</p>
<p>An equivalent process takes place in regions in which the root structure tapers off too much for large FoVs to correctly classify them. This situation is shown in <xref ref-type="fig" rid="f6">
<bold>Figures&#xa0;6K-O</bold>
</xref>. Starting with a FoV of edge size 5, a voxel classified as &#x2018;<italic>non-root</italic>&#x2019; is queued again with a smaller FoV of edge size 3, illustrated with an orange color-coding. Using the smaller FoV, the voxel is then correctly classified as <italic>&#x2018;root&#x2019;</italic> and the growing phase now continues along the thinner root structure. The further progressed state in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6O</bold>
</xref> shows that this effect can happen with multiple voxels at the same time. Note that the processing order of the queued voxels in these sketches do not follow the standard FIFO order of a queue, but rather the order was chosen to illustrate the idea. During the inference process, the queued voxels are evaluated simultaneously as part of the same batch.</p>
<p>As described above, the &#x2018;growing phase&#x2019; leaves the FoV size unchanged between the current voxel and the new neighbors which are pushed in the process queue. The FoVs edge size will only be decreased in areas, where a voxel has been classified as &#x2018;non root&#x2019;. This approach will carry the high confidence levels of the large FoVs through the already grown root structures and sacrifice it for a higher spatial accuracy only where needed. In this way, the radius of the FoV will decrease more and more until a given lower boundary value of <italic>r</italic> = 1 and edge size of 3 is reached (depicted with an orange color-coding). A classification as &#x2018;non root&#x2019; with this FoV, as indicated in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6O</bold>
</xref>, will lead to the next phase of the process.Following the downward &#x2018;yes&#x2019; arrow from diamond (b) in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, the algorithm now knows with the highest confidence and spatial accuracy, that the current voxel is <italic>not</italic> from the <italic>&#x2018;root&#x2019;</italic> class and marks it as such. However, this level of spatial accuracy cannot yet be assured for the previously classified root voxels. To tackle this problem, a so-called &#x2018;pruning phase&#x2019; is initiated in the affected local region. At this point, it is possible that the already grown structure extends over the true root&#x2019;s boundaries. In contrast to the premature edge detection in the growing phase, the gradients of the gray level values in the CT data can also promote a delayed edge detection for larger FoVs.</p>
<p>
<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6P</bold>
</xref> shows an example with a FoV of edge length <italic>l =</italic> 5 voxels. In <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6Q</bold>
</xref> the process queues the same voxel again with a smaller FoV with radius <italic>r</italic> = 1 and an edge length <italic>l =</italic> 3, which is the smallest FoV possible. As the voxel is already located considerably across the root boundary, the smaller FoV classifies the voxel as &#x2018;non root&#x2019; in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6R</bold>
</xref>. Since the smallest allowed FoV was already used, this initiates the pruning phase in that region of the volume, which allows to double check the currently proposed segmentation by the larger FoVs.</p>
<p>Under these circumstances it may happen that neighboring voxels, which have previously been classified as <italic>&#x2018;root&#x2019;</italic> within a larger FoV, will be re-evaluated using the smallest FoV in later iterations. If the re-evaluation reveals misclassifications by the previously used larger FoV, the pruning process is successful, and the root boundary is corrected and shifted back by updating the classification result at that voxel from &#x2018;<italic>root</italic>&#x2019; to &#x2018;<italic>non-root</italic>&#x2019;.</p>
<p>
<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6S</bold>
</xref> shows this effect of requeuing a voxel sample with a FoV of edge length <italic>l =</italic> 3, where previously a FoV of edge length <italic>l =</italic> 5 has predicted the class <italic>&#x2018;root&#x2019;</italic> in an earlier iteration. When the re-evaluation detects the opposing class, the process is repeated in such a way that it again follows the downward arrow in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> from diamond (b), resulting in another alternative loop.</p>
<p>Even though it only affects voxels, which were classified as <italic>&#x2018;root&#x2019;</italic> by a FoV larger than the smallest one, this path also results in a flood filling operation. Thereby, the boundary of the root is shifted slightly backwards to where the most spatially accurate FoV detects it to be. This is shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6T</bold>
</xref>, where in a slightly progressed state of the process, more voxels have been pruned and queued for re-evaluation. If the re-evaluation step confirms the previous result, the pruning phase terminates, and no new neighbors will be pushed to the queue. One example of this possibility is the lower neighbor of the originally discussed voxel in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6P</bold>
</xref>. If all voxels in that region reach this state, the segmentation of the local root structure can be considered complete.</p>
</sec>
<sec id="s6_3">
<label>6.3</label>
<title>Parallelization</title>
<p>Through the usage of the queue data structure, a parallel implementation of the inference evaluation processes as well as processes concerned with the enqueing of new neighbors is possible. Thus, multiple growing and pruning phases can be performed at the same time in different regions in the CT volume. This is also indicated in the lower right corner of <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6T</bold>
</xref>, where a new pruning phase has started on the other side of the root. When the queue is empty, the inference algorithm will terminate for the whole volume.</p>
</sec>
</sec>
<sec id="s7">
<label>7</label>
<title>Experiments and results</title>
<p>Using the data (Section 3) and the methods described above, a CNN was trained, using a SPP-layer (Section 4) to improve the scale invariance during the segmentation of root structures in CT data. A weakly supervised training (Section 5) loop was applied to improve noisy label data through the generalizing capabilities of neural networks. Using three training loops, three networks were trained from scratch, whose progress is presented in Section 7.1.</p>
<p>To create an adequate data set, a sub-labeling technique (Section 3.5) was applied, which categorizes each label into a more specific sub-label, by differentiating between content types and location in each training sample. This approach enables a fine-grained control over the diversity and difficulty of the training samples. Even though the impact of this sub-labeling is difficult to quantify, in Section 7.2 and Section 8.1, results are presented and discussed with respect to classifying the plant pot (included in the CT scans, see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>) which was frequently misclassified by networks trained <italic>without</italic> the sub-labeling technique.</p>
<p>For the testing phase, a novel inference approach was developed (Section 6), which operates in a flood-filling (or volume growing) scheme and can dynamically adapt the size of the field-of-view (FoV) of the input samples to achieve an overall high-quality segmentation. The results achieved with the proposed dynamic inference are compared with a na&#xef;ve (brute-force) inference algorithm as well as the reference <italic>RootForce</italic> segmentations (Section 3).</p>
<sec id="s7_1">
<label>7.1</label>
<title>Network training iterations</title>
<p>
<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> depicts the loss curves for the training (solid) and validation (dotted) data set over all three training iterations. A vast improvement from the first to the second iteration can be observed, which also corresponds to an improvement of the label data available for successive iterations. The improvement from the second to the third iteration is significantly reduced. Hence it can be concluded that the chosen net architecture (Section 4) has exhausted its ability to generalize based on the available small training data set (Section 3.2), as the training has reached a minimum. Hence, the training was stopped after the third iteration and the network was stored as &#x2018;final network&#x2019; <italic>N<sub>2</sub>
</italic>.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Showcase of the improvement, measured by the cross-entropy loss function, for the training (solid) and validation (dotted) curves over three weakly supervised training iterations. Curves show original data points; no smoothing was applied.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g007.tif"/>
</fig>
</sec>
<sec id="s7_2">
<label>7.2</label>
<title>Test setup</title>
<p>Each of the three evaluated inference methods (<italic>RootForce</italic>, na&#xef;ve inference, dynamic inference) apply an effective volume mask to reduce the region that needs processing; These are described here for reference:</p>
<list list-type="simple">
<list-item>
<p>(1) Firstly, <italic>RootForce</italic> automatically detects the <italic>&#x2018;air-soil&#x2019;</italic> transition at the top of the CT-volume and excludes all <italic>&#x2018;air&#x2019;</italic> voxels above from further assessment. Some of the CT&#x2019;s bottom layers are excluded as well, since the possible aggregation of water at the bottom of the pot causes higher attenuation coefficients and after the CT reconstruction, water aggregation yields CT gray level values similar to those of root structures. Hence, <italic>RootForce</italic> tends to misclassify humid soil as <italic>&#x2018;root&#x2019;</italic> if these layers would not be excluded. <italic>RootForce</italic> is able to estimates the maximal layer depth dynamically, which corresponds to the humid soil humidity.</p>
</list-item>
<list-item>
<p>(2) For the <italic>na</italic>&#xef;<italic>ve inference</italic> algorithm with the DCNN, the plant pot mask is applied, which was introduced in Section 3.5. Thereby, the classification of many background voxels was omitted, which saves approximately a third of the necessary computation time.</p>
</list-item>
<list-item>
<p>(3) For the proposed <italic>dynamic inference</italic>, only the masked bottom of the pot was excluded. As mentioned in <bold>Section 3.5</bold>, one goal was to develop and train a network which does not misclassify the plant pot voxels as <italic>&#x2018;root&#x2019;</italic> voxels. To test this, the test data volumes were inferred without any mask of the pots, and it was found that the mantle area of the pot is successfully absent from the segmentation. However, the bottom of the pot was falsely included in the segmentation every time a root branch reached down far enough. Hence, for the presented results the pot bottom was masked during inference to have a fair comparison with the other methods, as extra voxels from the pot bottom would overshadow the voxel count of the small roots and render a quantitative comparison more challenging. Furthermore, when visualizing the 3D structure (see <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>), the pot bottom practically eliminates the contrast needed to recognize small roots in the foreground.</p>
</list-item>
</list>
<p>Two exceptions were made for the dynamic inference results, which are marked by an asterisk in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> and <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, namely volumes <italic>V</italic>
<sub>13</sub> and <italic>V</italic>
<sub>14</sub>. Here the volumes were initially inferred with a mask that left the upper ridge of the pot bottom accessible to the flood-filling operation. This led to a segmented ring in <italic>V</italic>
<sub>14</sub>, where the upper ridge around the pot was falsely classified as <italic>&#x2018;root&#x2019;</italic>. This in turn allows the network to find additional small roots that otherwise would not have been connected to the main root-structure, as the mask of the pot acts as a natural barrier to the detection of root branches that first grow to the bottom of the pot and then <italic>reach back up</italic>. In these exceptions, the additionally detected small roots were preserved, while the ring artifacts belonging to the pot bottom were deleted in a post-processing step with another mask. By this approach, the ability of the proposed network to classify roots that were <italic>not</italic> detected by the reference method <italic>RootForce</italic> is better demonstrated. Moreover, these upward growing roots are a prominent example of the network correctly classifying the pot mantle as <italic>&#x2018;non-root&#x2019;</italic>, since long segments of those branches practically stick directly to the plant pot. While this results in a disconnected segmentation structure, it enables a more sensible comparison of the voxel counts.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Fused renderings of the segmented 3D root systems from the test data, obtained from <italic>RootForce</italic> and the proposed DCNN <italic>N<sub>2</sub>
</italic>. The two volumes marked with an asterisk &#x2217; had an additional post-processing step (see text). Colors decode the segmentation methods. Yellow: intersection between voxels segmented by the proposed DCNN with a 50% threshold and <italic>RootForce</italic>. Green: additionally detected root branches by the DCNN with a 50% threshold. Blue: further additionally detected root branches by the DCNN with a 20% threshold. Red: root segments only discovered by <italic>RootForce</italic>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g008.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison of root voxel counts for the test data from the reference method <italic>RootForce</italic> (S<sub>RF</sub>) and the proposed DCNN approach with thresholds of 50% and 20% on the class probabilities (S<sub>N50</sub>, S<sub>N20</sub>).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="left">Color</th>
<th valign="middle" rowspan="2" align="left">Domain</th>
<th valign="middle" colspan="9" align="center">Voxel counts</th>
</tr>
<tr>
<th valign="middle" align="right">V<sub>6</sub>
</th>
<th valign="middle" align="right">V<sub>7</sub>
</th>
<th valign="middle" align="right">V<sub>8</sub>
</th>
<th valign="middle" align="right">V<sub>9</sub>
</th>
<th valign="middle" align="right">V<sub>10</sub>
</th>
<th valign="middle" align="right">V<sub>11</sub>
</th>
<th valign="middle" align="right">V<sub>12</sub>
</th>
<th valign="middle" align="right">V<sub>13*</sub>
</th>
<th valign="middle" align="right">V<sub>14*</sub>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Red</td>
<td valign="middle" align="left">S<sub>RF</sub> \ S<sub>N50</sub>
</td>
<td valign="middle" align="right">90,904</td>
<td valign="middle" align="right">106,012</td>
<td valign="middle" align="right">111,695</td>
<td valign="middle" align="right">6,303</td>
<td valign="middle" align="right">86,306</td>
<td valign="middle" align="right">77,135</td>
<td valign="middle" align="right">74,401</td>
<td valign="middle" align="right">82,957</td>
<td valign="middle" align="right">235,468</td>
</tr>
<tr>
<td valign="middle" align="left">Yellow</td>
<td valign="middle" align="left">S<sub>RF</sub> &#x2229; S<sub>N50</sub>
</td>
<td valign="middle" align="right">3,091,392</td>
<td valign="middle" align="right">5,124,173</td>
<td valign="middle" align="right">6,743,898</td>
<td valign="middle" align="right">229,747</td>
<td valign="middle" align="right">4,460,024</td>
<td valign="middle" align="right">4,943,356</td>
<td valign="middle" align="right">1,845,211</td>
<td valign="middle" align="right">3,563,454</td>
<td valign="middle" align="right">4,973,605</td>
</tr>
<tr>
<td valign="middle" align="left">Green</td>
<td valign="middle" align="left">S<sub>N50</sub> \ S<sub>RF</sub>
</td>
<td valign="middle" align="right">79,988</td>
<td valign="middle" align="right">4,944,945</td>
<td valign="middle" align="right">130,176</td>
<td valign="middle" align="right">17,067</td>
<td valign="middle" align="right">198,783</td>
<td valign="middle" align="right">88,672</td>
<td valign="middle" align="right">88,683</td>
<td valign="middle" align="right">69,804</td>
<td valign="middle" align="right">500,967</td>
</tr>
<tr>
<td valign="middle" align="left">Blue</td>
<td valign="middle" align="left">S<sub>N20</sub> \ (S<sub>RF </sub> &#x222a; S<sub>N20</sub>)</td>
<td valign="middle" align="right">51,126</td>
<td valign="middle" align="right">2,213,364</td>
<td valign="middle" align="right">76,944</td>
<td valign="middle" align="right">16,357</td>
<td valign="middle" align="right">163,549</td>
<td valign="middle" align="right">53,569</td>
<td valign="middle" align="right">56,548</td>
<td valign="middle" align="right">92,750</td>
<td valign="middle" align="right">337,683</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The two volumes marked with an asterisk &#x2217; had an additional post-processing step (see text).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In contrast to the reference segmentation method <italic>RootForce</italic>, the proposed DCNN approach is able to detect and segment plant structures <italic>above</italic> ground. Nevertheless, the comparison has been restricted to root regions below ground.</p>
<p>For the initialization of the proposed dynamic inference algorithm, single seed voxels were used, which were a-priori known to be part of the <italic>&#x2018;root&#x2019;</italic> class. These positions are usually not known exactly, however, they can be found <italic>via</italic> thresholding within the main plant stem. Alternatively, a small set of voxels enclosing a region in the volume that is likely to contain a root can be used as seed voxels.</p>
</sec>
<sec id="s7_3">
<label>7.3</label>
<title>Visualization</title>
<p>In the following, the test data segmentations are presented with a visual comparison, based on a fused rendering of the 3D structure of the proposed dynamic inference method using <italic>N<sub>2</sub>
</italic> as DCNN approach and the reference method <italic>RootForce</italic> (see <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>). Additionally, the segmentation results of the dynamic inference are visually compared to the segmentation of the same network using a <italic>na&#xef;ve inference</italic> algorithm, as depicted in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Comparison of the DCNN <italic>N<sub>2</sub>
</italic> segmentation results from a na&#xef;ve inference approach and the proposed dynamic inference algorithm for volume <italic>V</italic>
<sub>6</sub>. <bold>(A)</bold> Na&#xef;ve inference with FoV <italic>r</italic> = 5, <bold>(B)</bold> na&#xef;ve inference with FoV <italic>r</italic> = 5 + CCA, <bold>(C)</bold> proposed novel dynamic inference. Color-coding is identical to <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>. Yellow: intersection between voxels segmented by the proposed DCNN with a 50% threshold and <italic>RootForce</italic>. Green: additionally detected root branches by the DCNN with a 50% threshold. Blue: further additionally detected root branches by the DCNN with a 20% threshold. Red: root segments only discovered by <italic>RootForce</italic>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1120189-g009.tif"/>
</fig>
<p>The affiliation between classified voxels and the underlying segmentation approach is denoted <italic>via</italic> set notation. Firstly, the set of all voxels segmented by the <italic>RootForce</italic> approach is denoted as <italic>S</italic>
<sub>RF</sub>. Secondly, the set of all voxels segmented by the proposed network <italic>N</italic>
<sub>2</sub> (from the third training iteration) and using a threshold <italic>&#x3b8;</italic> = 50% on the class probabilities (the network has a confidence above 50% that a voxel belongs to the <italic>&#x2018;root&#x2019;</italic> class) is referred to as <italic>S</italic>
<sub>N50</sub>. Finally, the network segmentation using a threshold of <italic>&#x3b8;</italic> = 20% is denoted by <italic>S</italic>
<sub>N20</sub>.</p>
<p>The voxel coloring scheme in <xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8</bold>
</xref>, <xref ref-type="fig" rid="f9">
<bold>9</bold>
</xref> decodes the segmentation method and is denoted in set notation in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, where the &#x2018;\&#x2019; sign denotes the difference of two sets (e.g. &#x2018;A\B&#x2019; translates to &#x2018;set A without set B&#x2019;): <italic>&#x2018;yellow&#x2019;</italic>, <italic>S</italic>
<sub>N50</sub> &#x2229; <italic>S</italic>
<sub>RF</sub> indicates the intersection between all voxels segmented by the proposed DCNN <italic>N<sub>2</sub>
</italic> and the voxels proposed by <italic>RootForce</italic>; <italic>&#x2018;green&#x2019;</italic>, <italic>S</italic>
<sub>N50</sub> \ <italic>S</italic>
<sub>RF</sub> denotes <italic>additionally</italic> detected root branches by the DCNN; <italic>&#x2018;blue&#x2019;, S</italic>
<sub>N20</sub> \ (<italic>S</italic>
<sub>N20</sub> &#x222a; <italic>S</italic>
<sub>RF</sub>) relates to further additionally detected thin root branches with the lower 20% threshold; <italic>&#x2018;red&#x2019;, S</italic>
<sub>RF</sub> \ <italic>S</italic>
<sub>N50</sub> indicates root segments, which are <italic>only</italic> discovered by <italic>RootForce</italic>. To obtain a quantitative comparison of the investigated methods, in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> the voxel counts for each color domain are listed separately.</p>
<p>For visualization purposes, the opacity of a one voxel thick layer around the yellow surface was set to zero in both <xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8</bold>
</xref>, <xref ref-type="fig" rid="f9">
<bold>9</bold>
</xref>. However, no root branches were deleted, and hence the qualitative extent and structure of the root system is not altered. This eliminates hollow red or green structures which wrap around the yellow domain and thereby reveals the true extent of the segmentations intersection. Nevertheless, this effect yields some apparent discrepancies in the voxel counts in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> when comparing it to the figures, as the voxel counts still include these hidden voxels. This effect is later discussed in Section 8.1. A similar procedure could have been applied to the surface of the blue regions. However, since the overlapping voxels are less significant in their count, and thus do not disturb the visualized information, this was neglected.</p>
</sec>
</sec>
<sec id="s8" sec-type="discussion">
<label>8</label>
<title>Discussion</title>
<p>In Section 8.1 the segmentation results on the test data set are compared between the proposed DCNN <italic>N<sub>2</sub>
</italic> segmentations and the analytical reference approach (<italic>RootForce</italic>). Differences between the na&#xef;ve and dynamic inference approaches are discussed and highlighted in Section 8.2. Performance considerations for the inference methods are outlines in Section 8.3. Finally, the shortcomings of the proposed methods and an outlook for possible improvements are presented in Section 8.4.</p>
<sec id="s8_1">
<label>8.1</label>
<title>Comparing to the reference</title>
<p>
<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> shows the segmented root systems of all test volumes (<italic>V</italic>
<sub>6</sub> &#x2013; <italic>V</italic>
<sub>14</sub>). Generally, in comparison the segmentation results of the proposed DCNN <italic>N<sub>2</sub>
</italic> approach and the analytical reference method (<italic>RootForce)</italic> are quite similar, and most of the root structures are detected by both algorithms (<italic>yellow</italic> voxels in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>). However, in all test volume the proposed deep neural network detected numerous additional thin root branches with the 50% threshold (<italic>green</italic> voxels in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>), and even more with the 20% threshold (<italic>blue</italic> voxels in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>), which have not been discovered with the reference approach <italic>RootForce</italic> (of which exclusively detected voxels are <italic>red</italic> in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>).</p>
<p>As one major finding, the <italic>quality</italic> of the root surface is increased using the proposed deep neural network approach. This effect is noticeable when focusing on the storage roots and comparing the count of different colored &#x2018;blobs&#x2019;, which are mostly misclassifications, attached to the root surface. For most of the test volumes, the red colored blobs outnumber the green and blue blobs. Thus, even with the low confidence threshold (<italic>&#x3b8;</italic> = 20%), the proposed DCNN approach yields competitive results.</p>
<p>The achieved voxel counts of detected root structures are listed in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> and can be used for a quantitative assessment of the different colored domains. The higher segmentation quality of the proposed approach is hinted at by an increased count of detected root voxels. However, there is a noticeable lack of visible, red-colored voxels in relation to visible green- and blue-colored voxels in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> and the comparatively small difference in voxel counts in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. This results from <italic>excluding</italic> voxels for an unbiased visual comparison, as explained in Section 7.3.</p>
<p>The excluded voxels of the red domain (voxels exclusively detected by <italic>RootForce</italic>) are concentrated along the surface of thin root branches, while <italic>excluded</italic> voxels of the green and blue domains (voxels exclusively detected by the DCNN) are scattered on the surface of large storage roots in similar quantity as the individual voxels which are still visible as dark dots on the yellow surface. Most voxels exclusively found by <italic>RootForce</italic> (<italic>red</italic>) widen the commonly detected (<italic>yellow</italic>) root branches, while voxels exclusively found by the proposed network approach (<italic>green, blue</italic>) extend the structure by new thin root branches. In terms of a qualitative comparison, there is a difference between detecting more surface voxels (like <italic>RootForce</italic>) versus detecting more root branches (like the DCNN). Contrary to the detection of wider branches, previously unknown root branches yield additional and desired information about the overall structure of the investigated plant root systems, especially when considering the ambiguity of &#x2018;correctly&#x2019; discretizing a continuous surface with voxels.</p>
<p>The most noticeable flaws of the proposed DCNN approach are the large, misclassified regions in the test volumes <italic>V</italic>
<sub>7</sub>, <italic>V</italic>
<sub>10</sub>, and <italic>V</italic>
<sub>14</sub>. These regions prohibit a useful comparison <italic>via</italic> voxel counts, since the misclassified voxels outnumber the ones that make up the small root branches, and hence cannot be easily separated from the true root system. Investigation of the cause for such errors showed an altered gray level profile in the affected areas, which was caused by higher water content in the soil and lead to an increase of the gray level values in the CT reconstruction. Unfortunately, these increased gray levels are very similar to the gray levels of many root structures. This observation indicates that the trained network has been fine-tuned to a specific soil mixture with a certain water saturation level which results in a characteristic gray level distribution adjacent to the plant. Hence, the obvious cause of this observed effect is the identical scanning parameters for the CT data acquisition of training data.</p>
<p>Focusing on the segmentation results of the DCNN with the lower threshold (<italic>&#x3b8;</italic> = 20%, &#x2018;blue&#x2019;), further significant improvements in detecting thin root branches can be observed. Using the lower threshold, the network detects the same root voxels as when using the higher threshold (<italic>&#x3b8;</italic> = 50%), and further extends the segmentation. Many of the test volumes (see <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>) have a similar amount of green- and blue-colored root branches, meaning using the lower threshold (<italic>&#x3b8;</italic> = 20%) resulted in nearly twice the amount of detected root voxels. A few small root branches are also detected by the reference method as well as the proposed network with <italic>&#x3b8;</italic> = 20%, but not with the 50% threshold. Two examples of this are in volume <italic>V</italic>
<sub>6</sub> (mainly red and partially blue branch, growing from the center to the lower right) and in volume <italic>V</italic>
<sub>12</sub> (furthest branch to the right, growing downwards).</p>
<p>Generally, the lower threshold (<italic>&#x3b8;</italic> = 20%) also comes with the cost of an increased count of misclassified voxels. However, the effect is more subtle than expected, since most misclassifications happen in regions, where the network already struggles with a higher threshold (<italic>&#x3b8;</italic> = 50%), e.g., at the humid sediment in volumes <italic>V</italic>
<sub>7</sub>, <italic>V</italic>
<sub>10</sub>, and <italic>V</italic>
<sub>14</sub>. Though, misclassifications of soil lumps sticking to the root surface are more frequent with lower threshold compared to <italic>RootForce.</italic>
</p>
<p>Some roots that follow the outer edge of the plant pot, as depicted for volumes <italic>V</italic>
<sub>8</sub>, <italic>V</italic>
<sub>13</sub> and <italic>V</italic>
<sub>14</sub>, highlight the success of the proposed and employed sub-labeling technique. In such situations, other networks models with the same network architecture and trained <italic>without</italic> the sub-labels, have frequently initiated the inference algorithm to follow the plant pot (by &#x2018;flood-filling&#x2019;) as well. This argument can also be made for the bottom of the plant pot as a counter example, since it was not focused on in training by the sub-labels (see Section 3.5). Note that the pot bottom was excluded from the segmentations in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> by a post-processing step as explained in the previous section.</p>
<p>It can be argued that the problem of the misclassification of the base of the pot shows the advantage of properly applying the proposed sub-labeling technique, as it allowed for exact control over samples from the pot mantle but not the pot bottom (see Section 3.5). More precisely, specifically choosing samples from the approximated segmentations of the pot mantle to augment the training data, resulted in successfully teaching the network to <italic>not</italic> recognize the <italic>&#x2018;pot&#x2019;</italic> voxels as <italic>&#x2018;root&#x2019;</italic> voxels. However, due to the complicated geometry of the pot bottom, no reference segmentations were available from which specific samples could augment the training data. Therefore, the alternative of randomly sampling from the bottom volume to statistically include challenging samples in the training data was used. Unfortunately, this did not focus learning on such samples sufficiently and hence was not successful in teaching the network to identify voxels from the pot bottom as &#x2018;<italic>non-root</italic>&#x2019;. This problem could be alleviated by possibly increasing the training samples from the pot bottom. However, this also could alter the proportion of multiple sub-labels to an unknown degree and therefore opposes the principle of fine-tuning the composition of the training data.</p>
</sec>
<sec id="s8_2">
<label>8.2</label>
<title>Comparing to the na&#xef;ve inference</title>
<p>As described in <bold>Section 6</bold>, the <italic>na&#xef;ve algorithm</italic> iterates through every voxel in the volume independently. This approach has the disadvantage that mostly &#x2018;<italic>non-root</italic>&#x2019; voxels are queried, since they constitute most voxels in a volume and thus, waste expensive computation time. Moreover, this approach is more vulnerable to misclassifications as the deep neural network never connects the queried results within one volume, leading to a cluttered 3D structure (see <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9A</bold>
</xref>). Here, a helpful post-processing step is the application of a &#x2018;connected component analysis&#x2019; (CCA), which extracts the largest detected structures and removes most of the clutter. In <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> the segmentations from the proposed DCNN <italic>N<sub>2</sub>
</italic> from the third training iteration are depicted, comparing the na&#xef;ve and dynamic inference methods. To aid the visualization, the reference segmentation (<italic>RootForce</italic>) is included in the illustrations for an identical color-coding to <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>. In <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9A</bold>
</xref>, the raw output of the na&#xef;ve inference algorithm is depicted for Volume <italic>V</italic>
<sub>6</sub>. It is obvious that there exists a lot of clutter, namely misclassifications, generated by the network (depicted as <italic>green</italic>- and <italic>blue</italic>-colored voxels), which obscure the view onto the root structure in the center.</p>
<p>
<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref> shows the result after applying the connected component analysis (CCA) which deletes all structures consisting of less than 1,000 connected voxels. This results in only some larger clumps of soil in the segmentation. Some large, misclassified structures are part of the pot (top left) and some topsoil sticking to the pot (top right, background). Also, small root structures that have been detected by the network, become visible in this view. Those are found around the root in the lower half of the illustration. The segmentation result of the proposed dynamic inference is depicted in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9C</bold>
</xref>. Since the proposed dynamic inference approach automatically generates only one single connected component, unwanted structures in the volume are automatically discarded. However, this comes at the cost of possibly undetected root sections, which <italic>could</italic> have been correctly classified by the network using the na&#xef;ve inference (cf. <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref>) but remain hidden for the flood-filling approach. For example, in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref> there are three thin green root branches visible in the foreground (horizontally centered). In <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9C</bold>
</xref> one of them is missing in the segmentation using a threshold of \theta = 50% but detected by the lower 20% threshold (colored in blue). Another branch (colored in red with some blue depicted in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9C</bold>
</xref>), indicates that only the reference method (<italic>RootForce</italic>) and the deep neural network with threshold <italic>&#x3b8;</italic> = 20% was able to detect it, while the same branch is colored <italic>yellow</italic> in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref>, indicating that the network with threshold <italic>&#x3b8;</italic> = 50% can detect it as well when it is not blocked off by the flood-filling process.</p>
<p>In conclusion, an obvious tuning parameter is <italic>&#x3b8;</italic>, thresholding the class probability of the network output. Lower thresholding (e.g., <italic>&#x3b8;</italic> = 20%) is able to segment thinner and finer roots, however, is also susceptible to more misclassifications. For the na&#xef;ve algorithm, this means also more unwanted large, connected components which are harder to separate from the desired root branches, as can be seen in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref> by the additional blue soil clumps. The segmentation of the dynamic algorithm is affected more positively by a low threshold, since it increases the connectivity of the root components as described above, while other parts of the segmentation are barely altered. This, on one hand, leads to more detected thin roots, which are connected to the main-root structure, but on the other hand, leads to a slightly rougher root surface on a few branches. This can be seen in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9C</bold>
</xref> by the thin blue root branches and the blue dots along the <italic>yellow</italic> surface respectively.</p>
<p>We also note that the na&#xef;ve inference only uses one FoV for all queries, while the proposed <italic>dynamic inference</italic> adaptively changes the radius <italic>r</italic> of the FoV. A FoV with edge length <italic>l =</italic> 5 was applied for the na&#xef;ve inference. Using larger FoVs with a constant radius would lead to less, but still obstructing, clutter in the raw segmentation and would furthermore fail to detect thin root branches. When investigating the effect of <italic>varied</italic> FoV sizes of the dynamic algorithm, it becomes apparent that most voxels are evaluated with the largest, initial FoV size. Adaptation to smaller sizes happens only at the direct root boundaries or very thin root structures with small local voxel counts. Hence, the effect is not statistically significant for the voxel count. However, using only larger FoVs with the flood-filling approach misses most of the thin roots and yields thicker root structures overall, due to the missing pruning step.</p>
<p>Furthermore, for the na&#xef;ve inference the count of root segments kept after the CCA is dependent on a hyper-parameter. This could be set in such a way that only the largest connected component, or components which fulfill known geometric requirements such as elongation restrictions, are kept in the volume. This could yield segmentations, which closer resemble the segmentation of the proposed dynamic algorithm. However, choosing adequate criteria to discard disconnected components becomes challenging since small and thin roots are easily caught erroneously when trying to declutter the segmented volume.</p>
<p>Some of the soil lumps in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref> can also show up within the dynamic inference, if its initialization consists of <italic>guessing</italic> where possible root voxels are located in the volume, e.g., by selecting a small surface of seed voxels in the volumes center. As long as the root structure crosses the surface, all parts of the roots will be detected as shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9C</bold>
</xref>. If some single soil lumps &#x2013; depicted in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref> &#x2013; happen to cross this surface, they would be included in the segmentation of the dynamic inference as well.</p>
</sec>
<sec id="s8_3">
<label>8.3</label>
<title>Performance</title>
<p>In the following, the evaluation times of the different inference methods are compared. Improvements are mainly achieved with respect to the na&#xef;ve method, as the <italic>RootForce</italic> algorithm is based on classic image processing methods whose run-time is highly optimized.</p>
<p>The <italic>RootForce</italic> algorithm always needs the same time duration to evaluate a volume of a fixed size. For our test volumes (1024 &#xd7; 1024 &#xd7; 900 voxels, see Section 3) this amounts to approximately 10 minutes on an Intel Xeon CPU (E5-2620v4@2.10GHz) with 256 GB system RAM.</p>
<p>Evaluation with the na&#xef;ve method also takes the same time for a fixed volume size. However, the run-time significantly depends on the used FoV size. The limiting factor is the VRAM used on the graphics card for each batch. For sequential processing on one GeForce GTX 1080 Ti GPU, conducted measurements with FoV edge lengths of <italic>l</italic> = 5 and <italic>l</italic> = 15, allowed for batches of size 2<sup>14</sup> = 16384 samples and 2<sup>9</sup> = 512 samples respectively, and resulted in evaluation times of roughly 120 and 3,000 minutes respectively.</p>
<p>The run-time of the dynamic inference algorithm is dependent on the amount of connected root voxels, which are present in the volume, whereby the connectivity is decisively dependent on the used threshold on the class probability. A more subtle influence comes from the structure of the root itself. Since more voxels are queued simultaneously during the flood-filling of thicker roots, the evaluation can leverage a better parallelization <italic>via</italic> the batch size. Contrary, in thin roots, voxels are queued more sequentially which results in smaller batches per query. For a maximal speedup, the parallel processing capabilities mentioned in Section 6.3 are utilized. With four GeForce GTX 1080 Ti GPUs (in one computation node), the smallest root system (<italic>V<sub>9</sub>
</italic>) can be segmented in 5 minutes, the largest root system (<italic>V<sub>8</sub>
</italic>) needs up to 25 minutes. Misclassifications as in <italic>V<sub>7</sub>
</italic> also cost extra time, which in this case amounted to 36 minutes.</p>
<p>For comparison we assume the best conditions for the na&#xef;ve algorithm, i.e., inference with FoV of edge length l = 5 and perfect speedup for parallelization (i.e., the measured processing time of 120 minutes is divided by the count of used GPUs). When using four GPUs, the na&#xef;ve algorithm can be quicker for exceptionally large root systems, however for our testing volumes this was not the case. When using more than four GPUs (in multiple computation nodes) the na&#xef;ve algorithm will surpass the dynamic algorithm, since the communication overhead for parallel processing of the na&#xef;ve algorithm is trivial and hence, has an advantageous scaling property.</p>
<p>However, when scaling with respect to volume size, the dynamic algorithm will have an advantageous scaling property, since the fraction of root voxels to volume size grows slower than the total voxel count in the volume. Therefore, less queries are required in the dynamic algorithm to process a large volume. Even more so, if no mask is available to generously reduce the processed region for the na&#xef;ve algorithm (see Section 7.2). Scaling with respect to volume size applies to larger objects or objects of the same size being scanned with a higher voxel resolution.</p>
</sec>
<sec id="s8_4">
<label>8.4</label>
<title>Shortcomings and outlook</title>
<p>One general complication of evaluating the quality of the proposed set of methods is the unobtainable, pristine ground truth data of the root structure. Even manual annotation would most likely not yield the desired quality, due to the error-prone fine structures and low contrast gray levels (see Section 3.1). Hence, the exact structure and total voxel count of the true root system is never known, and thus, it cannot be stated how many root branches are left undetected or are falsely segmented. Furthermore, it is challenging &#x2013; if not even impossible &#x2013; to decide whether undetected small roots are the result of a lack in generalization capabilities of the proposed deep network model or are due to too many false negative samples in the training data set or changes in the scanning system. Essentially, this effect blurs the line when deciding at which point the deep neural network starts overfitting to the true data distribution, as the network model possibly learns unknown characteristics of the flawed training data set, which can then not be called out by the validation and test data sets inheriting the same flaws. In this case, &#x2018;successfully&#x2019; learning generalizations of adequate features on the flawed data might be equivalent to a data set specific adaptation, i.e., overfitting, when considering the unknown ground truth data. Nevertheless, typical cues for overfitting do not show up on the available training curves for the data sets (see Section 7.1).</p>
<p>An obvious shortcoming of the investigated approach is the problem of the bottom of the plant pot. Hence, for future extensions and applications, it is advisable to use plant pots with either a more trivial geometry or obtaining an alternative segmentation of the plant pot for the sub-labels (see Section 3.5) and thus focus more on the negative samples during the training. Alternatively, if the range of future application is limited, using a segmentation of the plant pot as mask to prohibit network queries outside the sediment region is also a viable option, since using such a mask during the inference run has no disadvantages for the computation time (see Section 8.3).</p>
<p>The comparison between the proposed dynamic inference and the na&#xef;ve inference algorithm shows that inference by flood-filling can be a drawback if the root structure is not one single connected component as seen from a certain threshold. Basically, root branches which would be detected by the deep network model will not show up in the dynamic inference segmentation, when the branch is by accident disconnected from the main root structure. Increasing the connectivity of the detected root structures is equivalent to an increase in the network&#x2019;s fidelity, which requires a training data set of high quality. If increased computation times can be handled or tolerated, an alternative approach could be to use the na&#xef;ve inference with a conservative CCA as post-processing (see <bold>Section 8.2</bold>). The remaining components in the volume could be separated into soil lumps and root branches by a shallow learning model and basic features computed from the geometry of the connected region.</p>
<p>The most significant disadvantage of the proposed DCNN-based segmentation approach is the lack of diverse CT scans of roots. All described experiments were carried out on scans of the same (<italic>Cassava</italic>) plant type, grown in the same soil mixture and captured with the same scanning parameters and CT scanner (see <bold>Section 3</bold>). Alternating any of these influencing factors will considerably change the problem statement in the form of the gray level distributions. This would limit the current network model in scope and usefulness, as the discussion on the humidity levels in the sediment already indicated (see <bold>Section 8.1</bold>). Generalizing the proposed network at this level will require a careful composition of diverse CT scans for the training procedure, and most likely a larger network architecture. Other possible approaches could be to train a generative model, to map arbitrary CT scans of root systems to a common generalized latent space, in which a simple model like ours could be used for final segmentations.</p>
<p>A drawback of the dynamic inference algorithm is that it was specifically developed to work with single voxel predictions from the DCNN. This requires significantly more forward passes through the neural network to evaluate every voxel, than modern <italic>image-to-image</italic> DCNNs, like the U-Net (<xref ref-type="bibr" rid="B30">Smith et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B31">Soltaninejad et&#xa0;al., 2020</xref>), could achieve. Nevertheless, a trade-off between fast computation time (e.g., by using U-Net architectures) and carrying semantic context between individual predictions (e.g., by using flood-filling procedures) has to be made. A promising direction for future research was proposed by <xref ref-type="bibr" rid="B16">Januszewski et&#xa0;al. (2018)</xref> (and recently evaluated by <xref ref-type="bibr" rid="B13">Gruber et&#xa0;al. (2021)</xref>) who integrated the flood-filling approach directly into an image-to-image DCNNs framework called &#x201c;Flood-Filling-Networks&#x201d;.</p>
</sec>
</sec>
<sec id="s9" sec-type="conclusions">
<label>9</label>
<title>Conclusion</title>
<p>In this work, we presented a modified deep convolutional neural network architecture including a spatial pyramid pooling layer paired with a dynamic inference algorithm and a sub-labeling method, which combined are capable of segmenting 3D plant root-structures automatically in CT reconstruction volumes. An analytical segmentation algorithm was used as reference and baseline, whose outcome was also used for a weakly supervised training scheme of the proposed network model. By incorporating the spatial pyramid pooling layer in the network model, the detection and segmentation of arbitrarily sized root samples in different scales was enabled. Furthermore, the use of sub-labels was introduced in the training phase to extract the most essential samples from the available sparse training data. Thereby, the learning process can be guided to find a robust optimum. To infer large CT-volumes of plant roots within reasonable time scales, a dynamic inference algorithm was presented which operates in a flood-filling (or volume growing) manner and adapts the sample size of each query to yield a high-quality segmentation.</p>
<p>The achieved results of the novel root-structure segmentation approach were qualitatively (by visualization) and quantitatively (by voxel count) compared with the analytical reference segmentation algorithm as well as by comparing the proposed dynamic to a na&#xef;ve inference procedure. It was shown that the proposed novel delineation approach for root structures shows a significant improvement by segmenting previously hard to detect thin and fine root branches. However, some root branches are still left undetected with the applied weakly supervised learning approach, as not all real-world (&#x201c;in the wild&#x201d;) possibilities have been considered in the available training data yet. Nevertheless, the proposed approach shows the advantage that inference of new volumetric root data is applicable without pre- or post-processing steps and has an advantageous scaling property of computation time for larger volumes.</p>
<p>In summary we conclude that the proposed novel deep-learning based segmentation approach for root-structures in CT-volumes improves and extends the possibilities of an analytical reference method, by detecting much more fine and small root branches.</p>
</sec>
<sec id="s10" sec-type="data-availability">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: Some of the CT data sets of the roots can be obtained from the corresponding author SG. Requests to access these datasets should be directed to stefan.gerth@iis.fraunhofer.de.</p>
</sec>
<sec id="s11" sec-type="author-contributions">
<title>Author contributions</title>
<p>JA proposed the dynamic inference and sub-labelling, did the implementations and generated the balanced data set, conducted the training and inference experiments of the DCNN, and &#x2013; together with RG &#x2013; drafted, wrote, and finalized the manuscript, including the graphics. RG supervised the implementation of the DCNN, brought in ideas about 3D flood-filling, helped with the evaluation and interpretation of the experiments, and &#x2013; together with JA &#x2013; drafted, wrote, and finalized the manuscript, including the graphics. NW suggested the core concept for the deep learning approach, generated the label date using <italic>RootForce</italic>, and supervised the implementation of the DCNN in the first phase. JC organized and performed the CT scans of the plants, and provided knowledge about the <italic>RootForce</italic> approach. NU as an expert in CT-data analysis gave advice to the team about the CT-scanning, and the data interpretation, he supported the organization of the test specimens for the measurements. TW supported and guided the concept and did the final review and editing of the manuscript. SG provided the idea of the experiments combining deep learning with CT-based plant-root analysis, suggested <italic>RootForce</italic> to generate training data and was also deeply involved in its development. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec id="s12" sec-type="funding-information">
<title>Funding</title>
<p>This work was partially supported by the Bavarian Ministry of Economic Affairs, Regional Development and Energy through the Center for Analytics&#x2013;Data&#x2013;Applications (ADA-Center) within the framework of &#x2018;BAYERN DIGITAL II&#x2019; (20-3410-2-9-8) and the Bill and Melinda Gates Foundation for funding this research through the grant INV-008053 &#x2018;Metabolic Engineering of Carbon Pathways to Enhance Yield of Root and Tuber Crops&#x2019; provided to Professor Dr. Uwe Sonnewald.</p>
</sec>
<sec id="s13" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s14" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Klassen</surname> <given-names>T. N.</given-names>
</name>
<name>
<surname>Keyes</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Daly</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>D. L.</given-names>
</name>
<name>
<surname>Mavrogordato</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Imaging the interaction of roots and phosphate fertiliser granules using 4D X-ray tomography</article-title>. <source>Plant Soil</source> <volume>401</volume>, <fpage>125</fpage>&#x2013;<lpage>134</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11104-015-2425-5</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Atkinson</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Pound</surname> <given-names>M. P.</given-names>
</name>
<name>
<surname>Bennett</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Wells</surname> <given-names>D. M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Uncovering the hidden half of plants using new advances in root phenotyping</article-title>. <source>Curr. Opin. Biotechnol.</source> <volume>55</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.copbio.2018.06.002</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ayhan</surname> <given-names>C.B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Kwan, tree, shrub, and grass classification using only RGB images</article-title>. <source>Remote Sens.</source> <volume>12</volume> (<issue>8</issue>), <elocation-id>1333</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs12081333</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cicek</surname> <given-names>&#xd6;.</given-names>
</name>
<name>
<surname>Abdulkadir</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Lienkamp</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Brox</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ronneberger</surname> <given-names>O.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>3D U-net: Learning dense volumetric segmentation from sparse annotation</article-title>,&#x201d; in <source>Medical image computing and computer-assisted intervention &#x2013; MICCAI 2016</source> (<publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>424</fpage>&#x2013;<lpage>432</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-46723-8_49</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dannowski</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Block</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Fractal geometry and root system structures of heterogeneous plant communities</article-title>. <source>Plant Soil</source> <volume>272</volume>, <fpage>61</fpage>&#x2013;<lpage>76</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11104-004-3981-2</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Smet</surname> <given-names>I.</given-names>
</name>
<name>
<surname>White</surname> <given-names>P. J.</given-names>
</name>
<name>
<surname>Bengough</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Dupuy</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Parizot</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Casimiro</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>Analyzing lateral root development: how to move forward</article-title>. <source>Plant Cell.</source> <volume>24</volume> (<issue>1</issue>), <fpage>15</fpage>&#x2013;<lpage>20</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1105/tpc.111.094292</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ferreira</surname> <given-names>T. R.</given-names>
</name>
<name>
<surname>C&#xe1;ssaro</surname> <given-names>F. A. M.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Pires</surname> <given-names>L. F.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>X-Ray computed tomography image processing &amp; segmentation: A case study applying machine learning and deep learning-based strategies</article-title>,&#x201d; in <source>X-Ray imaging of the soil porous architecture</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Mooney</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Young</surname> <given-names>I. M.</given-names>
</name>
<name>
<surname>Heck</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Peth</surname> <given-names>S.</given-names>
</name>
</person-group> (<publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-031-12176-0_5</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flavel</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Guppy</surname> <given-names>C. N.</given-names>
</name>
<name>
<surname>Rabbi</surname> <given-names>S. M. R.</given-names>
</name>
<name>
<surname>Young</surname> <given-names>I. M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>An image processing and analysis tool for identifying and analysing complex plant root systems in 3D soil using non-destructive analysis: Root1</article-title>. <source>PloS One</source> <volume>12</volume> (<issue>5</issue>), <fpage>1</fpage>&#x2013;<lpage>18</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0176433</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flavel</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Guppy</surname> <given-names>C. N.</given-names>
</name>
<name>
<surname>Tighe</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Watt</surname> <given-names>M.</given-names>
</name>
<name>
<surname>McNeill</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Young</surname> <given-names>I. M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Non-destructive quantification of cereal roots in soil using high resolution X-ray tomography</article-title>. <source>J. Exp. Bot.</source> <volume>63</volume> (<issue>7</issue>), <fpage>2503</fpage>&#x2013;<lpage>2511</lpage>. doi: <pub-id pub-id-type="doi">10.1093/jxb/err421</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Frangi</surname> <given-names>A. F.</given-names>
</name>
<name>
<surname>Niessen</surname> <given-names>W. J.</given-names>
</name>
<name>
<surname>Vincken</surname> <given-names>K. L.</given-names>
</name>
<name>
<surname>Viergever</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>1998</year>). &#x201c;<article-title>Multiscale vessel enhancement filtering</article-title>,&#x201d; in <source>Medical image computing and computer-assisted intervention &#x2014; MICCAI&#x2019;98</source> (<publisher-loc>Cambridge, MA, USA</publisher-loc>: <publisher-name>Springer Berlin Heidelberg</publisher-name>), <fpage>130</fpage>&#x2013;<lpage>137</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/BFb0056195</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Schl&#xfc;ter</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Blaser</surname> <given-names>S. R. G. A.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Vetterlein</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A shape-based method for automatic and rapid segmentation of roots in soil from X-ray computed tomography images: Rootine</article-title>. <source>Plant Soil</source> <volume>441</volume>, <fpage>643</fpage>&#x2013;<lpage>655</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11104-019-04053-6</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gerth</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Clau&#xdf;en</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Eggert</surname> <given-names>A.</given-names>
</name>
<name>
<surname>W&#xf6;rlein</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Waininger</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Wittenberg</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Semiautomated 3D root segmentation and evaluation based on X-ray CT imagery</article-title>. <source>Plant Phenomics</source> <volume>2021</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/2021/8747930</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gruber</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Gerth</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Clau&#xdf;en</surname> <given-names>J.</given-names>
</name>
<name>
<surname>W&#xf6;rlein</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Uhlmann</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Wittenberg</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Exploring flood filling networks for instance segmentation of XXL-volumetric and bulk material CT data</article-title>. <source>J. Nondestructive Eval.</source> <volume>40</volume> (<issue>1</issue>), <fpage>1</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10921-020-00734-w</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Spatial pyramid pooling in deep convolutional networks for visual recognition</article-title>,&#x201d; in <source>Computer vision &#x2013; ECCV 2014. lecture notes in computer science</source>, vol. <volume>8691</volume> Eds. <person-group person-group-type="editor">
<name>
<surname>Fleet</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Pajdla</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Schiele</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Tuytelaars</surname> <given-names>T.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-10578-9_23</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Pleiss</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hopcroft</surname> <given-names>J. E.</given-names>
</name>
<name>
<surname>Weinberger</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Snapshot ensembles: Train 1, get m for free</article-title>. <source>ArXiv</source> <fpage>arXiv.1704.00109</fpage>. doi: <pub-id pub-id-type="doi">
10.48550/arXiv.1704.00109</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Januszewski</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kornfeld</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>P. H.</given-names>
</name>
<name>
<surname>Pope</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Blakely</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Lindsey</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>High-precision automated reconstruction of neurons with flood-filling networks</article-title>. <source>Nat. Methods</source> <volume>15</volume>, <fpage>605</fpage>&#x2013;<lpage>610</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41592-018-0049-4</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Khoreva</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Benenson</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hosang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hein</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Schiele</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Simple does it: Weakly supervised instance and semantic segmentation</article-title>,&#x201d; in <conf-name>Proc&#x2019;s 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <publisher-loc>Honolulu, HI, USA</publisher-loc>, pp <fpage>1665</fpage>&#x2013;<lpage>1674</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR.2017.181</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krizhevsky</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>G. E.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>ImageNet classification with deep convolutional neural networks</article-title>. <source>Commun. ACM</source> <volume>60</volume> (<issue>6</issue>), <fpage>84</fpage>&#x2013;<lpage>90</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3065386</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Teng</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Recent advances in methods for in situ root phenotyping</article-title>. <source>Peer J.</source> <volume>10</volume>, <elocation-id>e13638</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.7717/peerj.13638</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Loshchilov</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Hutter</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>SGD-r: Stochastic gradient descent with restarts</article-title>. <source>ArXiv</source>, <fpage>abs/1608.03983</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1608.03983</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lucas</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vetterlein</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>X-Ray imaging of root&#x2013;soil interactions</article-title>,&#x201d; in <source>X-Ray imaging of the soil porous architecture</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Mooney</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Young</surname> <given-names>I. M.</given-names>
</name>
<name>
<surname>Heck</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Peth</surname> <given-names>S.</given-names>
</name>
</person-group> (<publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-031-12176-0_9</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mairhofer</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sturrock</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Bennett</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Mooney</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Pridmore</surname> <given-names>T. P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Extracting multiple interacting root systems using X-ray microcomputed tomography</article-title>. <source>Plant J.</source> <volume>84</volume> (<issue>5</issue>), <fpage>1034</fpage>&#x2013;<lpage>1043</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/tpj.13047</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mairhofer</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zappala</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Tracy</surname> <given-names>S. R.</given-names>
</name>
<name>
<surname>Sturrock</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Bennett</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Mooney</surname> <given-names>S. J.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>RooTrak: Automated recovery of three-dimensional plant root architecture in soil from X-ray microcomputed tomography images using visual tracking</article-title>. <source>Plant Physiol.</source> <volume>158</volume> (<issue>2</issue>), <fpage>561</fpage>&#x2013;<lpage>569</lpage>. doi: <pub-id pub-id-type="doi">10.1104/pp.111.186221</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Metzner</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Eggert</surname> <given-names>A.</given-names>
</name>
<name>
<surname>van Dusschoten</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Pflugfelder</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Gerth</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Schurr</surname> <given-names>U.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>Direct comparison of MRI and X-ray CT technologies for 3D imaging of root systems in soil: potential and challenges for root trait quantification</article-title>. <source>Plant Methods</source> <volume>11</volume>, <fpage>17</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-015-0060-z</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mooney</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Pridmore</surname> <given-names>T. P.</given-names>
</name>
<name>
<surname>Helliwell</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Bennett</surname> <given-names>M. J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Developing X-ray computed tomography to non-invasively image 3-d root systems architecture in soil</article-title>. <source>Plant Soil</source> <volume>352</volume>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11104-011-1039-9</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Phalempin</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lippold</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Vetterlein</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Schl&#xfc;ter</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>An improved method for the segmentation of roots from X-ray computed tomography 3D images: Rootine v.2</article-title>. <source>Plant Methods</source> <volume>17</volume>, <fpage>39</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-021-00735-4</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ronneberger</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Fischer</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Brox</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>U-Net: Convolutional networks for biomedical image segmentation</article-title>,&#x201d; in <source>Lecture notes in computer science</source> (<publisher-loc>Munich, Germany</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>234</fpage>&#x2013;<lpage>241</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-24574-4_28</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmidhuber</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning in neural networks: An overview</article-title>. <source>Neural Networks</source> <volume>61</volume>, <fpage>85</fpage>&#x2013;<lpage>117</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neunet.2014.09.003</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Kang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Shao</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>High throughput <italic>in situ</italic> root image segmentation based on the improved DeepLabv3+ method</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2020.576791</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smith</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Petersen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Selvan</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Rasmussen</surname> <given-names>C. R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Segmentation of roots in soil with U-net</article-title>. <source>Plant Methods</source> <volume>16</volume> (<issue>1</issue>), <fpage>13</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-020-0563-0</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soltaninejad</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sturrock</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Griffiths</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Pridmore</surname> <given-names>T. P.</given-names>
</name>
<name>
<surname>Pound</surname> <given-names>M. P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Three dimensional root CT segmentation using multi-resolution encoder-decoder networks</article-title>. <source>IEEE Trans. Image Process</source> <volume>29</volume>, <fpage>6667</fpage>&#x2013;<lpage>6679</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2020.2992893</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srivastava</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Krizhevsky</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Salakhutdinov</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Dropout: A simple way to prevent neural networks from overfitting</article-title>. <source>J. Mach. Learn. Res.</source> <volume>15</volume> (<issue>56</issue>), <fpage>1929</fpage>&#x2013;<lpage>1958</lpage>. doi: <pub-id pub-id-type="doi">10.5555/2627435.2670313</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thesma</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Mohammadpour Velni</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Plant root phenotyping using deep conditional GANs and binary semantic segmentation</article-title>. <source>Sensors</source> <volume>23</volume> (<issue>1</issue>), <elocation-id>309</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s23010309</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trachsel</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kaeppler</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>K. M.</given-names>
</name>
<name>
<surname>Lynch</surname> <given-names>J. P.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Shovelomics: high throughput phenotyping of maize (Zea mays l.) root architecture in the field</article-title>. <source>Plant Soil</source> <volume>341</volume>, <fpage>75</fpage>&#x2013;<lpage>87</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11104-010-0623-8</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Valdes</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Clarke</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Existing and potential statistical and computational approaches for the analysis of 3D CT images of plant roots</article-title>. <source>Agronomy</source> <volume>8</volume> (<issue>5</issue>), <fpage>71</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agronomy8050071</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wandel</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Landl</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Schnepf</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Behnke</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <conf-name>28th European Symposium on Artificial Neural Networks, Computational Intelligence and Machine Learning (ESANN 2020)</conf-name>; <conf-date>2020 October 2-4</conf-date>; Bruges, Belgium (2020). p. <fpage>463</fpage>&#x2013;<lpage>468</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>