<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="methods-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2023.1084778</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>EmergeNet: A novel deep-learning based ensemble segmentation model for emergence timing detection of coleoptile</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Das</surname>
<given-names>Aankit</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2078504"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Das Choudhury</surname>
<given-names>Sruti</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/641888"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Das</surname>
<given-names>Amit Kumar</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Samal</surname>
<given-names>Ashok</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/719520"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Awada</surname>
<given-names>Tala</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Institute of Radio Physics and Electronics, University of Calcutta</institution>, <addr-line>Kolkata, West Bengal</addr-line>, <country>India</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Computing, University of Nebraska-Lincoln</institution>, <addr-line>Lincoln, NE</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Natural Resources University of Nebraska-Lincoln</institution>, <addr-line>Lincoln, NE</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Computer Science and Engineering, Institute of Engineering and Management</institution>, <addr-line>Kolkata, West Bengal</addr-line>, <country>India</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Agricultural Research Division, University of Nebraska-Lincoln</institution>, <addr-line>Lincoln, NE</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Zhanyou Xu, Agricultural Research Service (USDA), United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Marcin Wozniak, Silesian University of Technology, Poland; Li Chaorong, Yibin University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Sruti Das Choudhury, <email xlink:href="mailto:s.d.choudhury@unl.edu">s.d.choudhury@unl.edu</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Technical Advances in Plant Science, a section of the journal Frontiers in Plant Science</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>02</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1084778</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>10</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>01</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Das, Das Choudhury, Das, Samal and Awada</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Das, Das Choudhury, Das, Samal and Awada</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The emergence timing of a plant, i.e., the time at which the plant is first visible from the surface of the soil, is an important phenotypic event and is an indicator of the successful establishment and growth of a plant. The paper introduces a novel deep-learning based model called EmergeNet with a customized loss function that adapts to plant growth for coleoptile (a rigid plant tissue that encloses the first leaves of a seedling) emergence timing detection. It can also track its growth from a time-lapse sequence of images with cluttered backgrounds and extreme variations in illumination. EmergeNet is a novel ensemble segmentation model that integrates three different but promising networks, namely, SEResNet, InceptionV3, and VGG19, in the encoder part of its base model, which is the UNet model. EmergeNet can correctly detect the coleoptile at its first emergence when it is tiny and therefore barely visible on the soil surface. The performance of EmergeNet is evaluated using a benchmark dataset called the University of Nebraska-Lincoln Maize Emergence Dataset (UNL-MED). It contains top-view time-lapse images of maize coleoptiles starting before the occurrence of their emergence and continuing until they are about one inch tall. EmergeNet detects the emergence timing with 100% accuracy compared with human-annotated ground-truth. Furthermore, it significantly outperforms UNet by generating very high-quality segmented masks of the coleoptiles in both natural light and dark environmental conditions.</p>
</abstract>
<kwd-group>
<kwd>event-based plant phenotyping</kwd>
<kwd>deep-learning</kwd>
<kwd>ensemble segmentation</kwd>
<kwd>emergence time detection</kwd>
<kwd>benchmark dataset</kwd>
</kwd-group>
<counts>
<fig-count count="12"/>
<table-count count="3"/>
<equation-count count="14"/>
<ref-count count="32"/>
<page-count count="15"/>
<word-count count="6822"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Image-based plant phenotyping has the potential to transform the field of agriculture through the automated measurements of phenotypic expressions, i.e., observable biophysical traits of a plant as a result of complex interactions between genetics and environmental conditions. Accurate computation of meaningful phenotypes contributes to the study of high yield of better-quality crops with minimum resources (<xref ref-type="bibr" rid="B6">Das Choudhury et&#xa0;al., 2018</xref>). A plant&#x2019;s phenome is defined as its observable characteristics or traits and is determined by the complex interaction between genotype and the environment. Plant phenotyping analysis has been an active research field for some time that adds to the understanding of yield and resource acquisition, and therefore, accelerates breedingcycles, improves our understanding of plant responses to environmental stresses, and contributes to global food security under changing climate. Image-based plant phenotypes can be broadly classified into three categories: structural, physiological, and event-based (<xref ref-type="bibr" rid="B8">Das Choudhury et&#xa0;al., 2019</xref>). The structural phenotypes characterize a plant&#x2019;s morphology, whereas physiological phenotypes refer to the physiological processes that regulate plant growth and metabolism (<xref ref-type="bibr" rid="B7">Das Choudhury and Samal, 2020</xref>).</p>
<p>The timing detection of important events in a plant&#x2019;s life cycle, for example, the emergence of coleoptile (i.e., protective sheath covering the emerging shoot) and new leaves, flowering, and fruiting, from time-lapse sequences has recently drawn significant research attention. Such phenotypes are called event-based phenotypes and provide crucial information in understanding the plant&#x2019;s vigor, which varies with the interaction between genotype and environment. While interest in event-based phenotyping forleaves, flowers, and fruits has increased substantially in recent times (<xref ref-type="bibr" rid="B29">Wang et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B4">Bashyam et&#xa0;al., 2021</xref>), detecting the emergence and monitoring of the growth of the coleoptile based on computer vision and artificial intelligence techniques is a budding research field with vast opportunities for exploration. Emergence is a significant phenotype that not only helps determine the dormancy of seeds for different genotypes in different environmental conditions but also various aspects of early plant growth stages.</p>
<p>Unlike the visual tracking of rigid bodies, for instance, vehicles and pedestrians, the emergence timing detection of living organs and tracking their growth over time requires a different problem formulation with an entirely new set of challenges. Firstly, the state-of-the-art rigid body object detection and tracking methods deal with objects of considerably larger size that do not change in shape and appearance during the period of consideration. In contrast, our problem is to detect the coleoptile at emergence, when it is tiny in appearance, and track its dynamics as leaves emerge and grow into a seedling. The growth monitoring of size and shape is obtained as a by-product of an ensemble segmentation technique that segments the coleoptile with high accuracy. Secondly, the background (soil) in typical emergence detection in high-throughput plant phenotyping systems is significantly more complex than the state-of-the-art visual tracking applications. The soil substrate is multicolored due to the presence of perlite and vermiculite which makes the background cluttered, rendering the detection of a tiny coleoptile extremely challenging. Finally, the images are captured for a longer time than visual tracking, typically days, in a greenhouse with natural and artificial lighting conditions resulting in significant illumination variations.</p>
<p>The central contribution of this paper is to introduce a novel ensemble segmentation model tailored to the detection and growth monitoring of living organs in cluttered backgrounds and illumination variations for applications in event-based plant phenotyping. EmergeNet, characterized by its custom-designed loss function, uses a novel weighted ensemble learning technique to minimize the variance of the predicted masks and the generalization error for emergence timing detection. A benchmark dataset is indispensable for the development of the algorithm and performance comparison. Therefore, we have developed a publicly available benchmark dataset called the University of Nebraska-Lincoln Maize Emergence Dataset (UNL-MED) consisting of time-lapse image sequences of maize coleoptiles under the aforementioned conditions.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related works</title>
<p>Multiple object tracking is challenging, yet it is of fundamental importance for many real-life practical applications (<xref ref-type="bibr" rid="B30">Xing et&#xa0;al., 2011</xref>). The survey paper by (<xref ref-type="bibr" rid="B10">Dhaka et&#xa0;al., 2021</xref>) provided comparisons of various convolutional neural networks and optimization techniques that are applied to predict plant diseases from leaf images. A comprehensive survey of multiple object tracking methods based on deep-learning is provided by (<xref ref-type="bibr" rid="B31">Xu et&#xa0;al., 2019</xref>). The method in (<xref ref-type="bibr" rid="B30">Xing et&#xa0;al., 2011</xref>) uses a progressive observation model followed by a dual-mode two-way Bayesian inference-based tracking strategy to track multiple highly interactive players with an abrupt view and pose variations in different sports videos, e.g., football, basketball, as well as hockey. A. Yilmaz et&#xa0;al (<xref ref-type="bibr" rid="B32">Yilmaz et&#xa0;al., 2006</xref>) showed that a plethora of research had been done in the field of object detection and tracking using various methods, including deep-learning algorithms. The method by (<xref ref-type="bibr" rid="B2">Aggarwal and Cai, 1999</xref>) gave an overview of the tasks involved in the motion analysis of a human body. (<xref ref-type="bibr" rid="B12">Doll&#xe1;r et&#xa0;al., 2009</xref>) worked on pedestrian detection, a key problem in computer vision, and proposed improved evaluation metrics. Computer vision based vehicle detection and tracking play an important role in the intelligent transport system (<xref ref-type="bibr" rid="B22">Min et&#xa0;al., 2018</xref>). The method in (<xref ref-type="bibr" rid="B22">Min et&#xa0;al., 2018</xref>) presents an improved ViBe for accurate detection of vehicles and uses two classifiers, i.e., support vector machine and convolutional neural network, to track vehicles in the presence of occlusions.</p>
<p>However, the use of deep neural networks for event-based plant phenotyping is in the early stage of research. The MangoYOLO algorithm (<xref ref-type="bibr" rid="B29">Wang et&#xa0;al., 2019</xref>) uses the YOLO object detector for detecting, tracking, and counting mangoes from a time-lapse video sequence. The method uses the Hungarian algorithm to correlate fruits between neighboring frames and a Kalman filter to predict the position of fruits in the following frames. A method for plant emergence detection and growth monitoring of the coleoptile based on adaptive hierarchical segmentation and optical flow using spatio-temporal image sequence analysis is presented in (<xref ref-type="bibr" rid="B1">Agarwal, 2017</xref>). A notable study in this domain includes the detection of budding and bifurcation events from 4D point clouds using a forward-backward analysis framework (<xref ref-type="bibr" rid="B18">Li et al., 2013</xref>). For a large-scale phenotypic experiment, the seeds are usually sown in smaller pots until germination and then transplanted to bigger pots based on a visual inspection of the germination date, size, and health of the seedlings. The method by (<xref ref-type="bibr" rid="B24">Scharr et al., 2020</xref>) developed an image-based automated germination detection system based on transfer-learning deep neural networks equipped with a visual support system for inspecting and transplanting seedlings. Deep-learning based ensemble segmentation technique has been recently introduced in medical image processing in ratio-based sampling for the arteries and veins in abdominal CT scans (<xref ref-type="bibr" rid="B14">Golla et&#xa0;al., 2020</xref>), skin lesion diagnosis using dermoscopic images (<xref ref-type="bibr" rid="B3">Arulmurugan et&#xa0;al., 2021</xref>), and portrait segmentation for application in surveillance systems (<xref ref-type="bibr" rid="B16">Kim et&#xa0;al., 2021</xref>).</p>
<p>To the best of our knowledge, there is no previous research that accurately detects the emergence timing of seedlings from a cluttered soil background and tracks its growth over a time-lapse sequence under extreme variations in illuminations using deep-learning based ensemble segmentation with custom loss functions. This paper proposes a novel algorithm that not only detects the emergence of the coleoptile under all the aforementioned challenging conditions but also successfully tracks its growth by creating an overlay mask on the image sequence even under extremely low light conditions at night. The proposed model, EmergeNet, uses deep-learning algorithms to create a novel segmentation model which can predict segmentation masks with high accuracy. We also release a benchmark dataset with ground-truth called UNL-MED, consisting of 3832 high-definition time-lapse image sequences of the maize coleoptiles.</p>
</sec>
<sec id="s3" sec-type="materials|methods">
<label>3</label>
<title>Materials and methods</title>
<sec id="s3_1">
<label>3.1</label>
<title>Dataset description</title>
<p>Benchmark datasets are critical in developing new algorithms and performing uniform comparisons among state-of-the-art algorithms. Hence, we created a benchmark dataset called the UNL-MED. It is organized into two folders, namely, &#x2018;Dataset&#x2019; and &#x2018;Training&#x2019;. The &#x2018;Dataset&#x2019; folder contains all of the 3832 raw high-definition images of resolution 5184 &#xd7; 3456. The &#x2018;Training&#x2019; folder contains two subfolders, namely, &#x2018;images&#x2019;, which has randomly selected 130 images for training, and &#x2018;masks&#x2019;, which contains 130 corresponding masks but downsampled to a resolution of 256 &#xd7; 256. The images are captured at an interval of two minutes under various external conditions, including varying illumination, cluttered background, warm and cool tone, starting from before the emergence occurred until the coleoptile is about 1 inch high to facilitate its growth monitoring. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> is used to demonstrate an example of extreme contrast of illumination of the images used in the experiment based on histogram analysis. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref> shows one of the brightest images and its histogram, whereas <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref> shows one of the darkest images and its corresponding histogram. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref> shows the darkest image after histogram equalization and its histogram. Each image contains nine pots sowed with maize seeds of different genotypes. A visible light camera fitted with a tripod was placed directly above the nursery to capture high-definition top-view images of all nine pots every two minutes. The dataset can be freely downloaded from <uri xlink:href="https://plantvision.unl.edu/dataset">https://plantvision.unl.edu/dataset</uri>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>
<bold>(A)</bold> An example of the brightest image from UNL-MED and its corresponding histogram; <bold>(B)</bold> an example of the darkest image from UNL-MED (the pots are marked in green circles) and its corresponding histogram; and <bold>(C)</bold> the darkest image after histogram equalization and its corresponding histogram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g001.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Dataset pre-processing</title>
<p>One of the most challenging and tedious tasks in image segmentation using deep-learning is the generation of ground-truth. In our case, it is in the form of binary masks corresponding to the plants in the images. For a custom dataset like the one used in this work, it is imperative that the masks are generated accurately, and therefore, it needs to be done manually. Utmost care has been taken while generating these masks as these serve as monitoring information during the semantic segmentation training to provide feedback to the neural network. In our experiment, we made use of the open-source manual annotation software developed by Visual Geometry Group (VGG) (<xref ref-type="bibr" rid="B13">Dutta et&#xa0;al., 2016</xref>). The flowchart of the data pre-processing is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Illustration of mask generation process for UNL-MED.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g002.tif"/>
</fig>
<p>From this figure, we can see that each image is first hand-annotated, and then the corresponding data is exported in &#x2018;JSON&#x2019; format for further processing. For each image, one or more masks are created from the exported data, and then they are superimposed to create the binary mask of the image. The images, along with their corresponding masks, are then fed into EmergeNet for training.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Proposed method: EmergeNet</title>
<p>In this section, we discuss the proposed model and its constituent parts in detail. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> represents the block diagram of the proposed method.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Flowchart of the proposed method.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g003.tif"/>
</fig>
<p>The first step is to generate masks from raw input images and pre-process them for training and evaluation. The images and their corresponding pixel labels are partitioned for training and testing. The training dataset is augmented to reduce overfitting and then fed to the model, EmergeNet, for training while the test dataset is used for evaluating the performance of the model. It can then be used to predict the emergence time of an image sequence. EmergeNet is a custom-made ensemble segmentation model that uses a weighted combination of loss functions and is specifically designed to detect tiny coleoptiles at the time of their first emergence from the soil under challenging conditions. It can also be used for growth monitoring of the plant over a time-lapse image sequence. EmergeNet consists of three underlying backbone architectures. The following subsections discuss these three standard backbone models, the EmergeNet Loss function, and the ensembling technique.</p>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>The backbone architectures</title>
<p>EmergeNet is built by ensembling three pre-trained networks, each based on the UNet architecture but with modified backbone models. It uses a custom-made loss function as well. The three backbone models used are SEResNet, InceptionV3, and VGG19.</p>
<sec id="s3_3_1_1">
<label>3.3.1.1</label>
<title>The UNet architecture</title>
<p>The UNet architecture, which is an extension of an encoder-decoder convolutional network, is known for its precise segmentations using fewer training images. Therefore, it is only logical to make optimum utilization of the UNet architecture for the task of fine-grain semantic segmentation. The basic intuition behind UNet is to encode the image, passing it through a convolutional neural network as it gets downsampled, and then decode it back, or upsample it to obtain the segmentation mask. However, it is experimentally found that using a pre-trained model as its encoder and decoder, rather than using the standard UNet architecture, the performance of the model improves significantly (<xref ref-type="bibr" rid="B17">Lagree et&#xa0;al., 2021</xref>).</p>
</sec>
<sec id="s3_3_1_2">
<label>3.3.1.2</label>
<title>UNet with SEResNet backbone</title>
<p>A novel architectural unit called the squeeze-and-excitation (SE) block has been introduced in (<xref ref-type="bibr" rid="B15">Hu et&#xa0;al., 2018</xref>). It adaptively recalibrates channel-wise feature responses by explicitly modeling interdependencies between channels at almost no computational cost. This is achieved by mapping the input to the feature maps for any given transformation. A detailed description of the structure of SE block and its operational characteristics are provided in (<xref ref-type="bibr" rid="B15">Hu et&#xa0;al., 2018</xref>). As an example, adding SE blocks to ResNet50 results in almost the same accuracy as ResNet101, but at a much lower computational complexity.</p>
</sec>
<sec id="s3_3_1_3">
<label>3.3.1.3</label>
<title>UNet with InceptionV3 backbone</title>
<p>The Inception architecture, unlike conventional convolutional networks, is a very complex, heavily engineered, deep neural network that uses filters of multiple sizes operating at the same level, rather than stacked convolutional layers. It enhances the utilization of available computational resources as well as improves performance significantly. The main idea of the Inception architecture is to find out how an optimal local sparse structure in a convolutional vision network can be approximated and covered by readily available dense components. InceptionV3 makes several improvements over earlier versions by including the following features: (a) <italic>Label smoothing</italic>, which is a regularization technique designed to tackle the problem of overfitting as well as overconfidence in deep neural networks; (b) <italic>Factorizing convolution</italic> to reduce the number of connections/parameters without decreasing the network efficiency; and (c) <italic>Auxiliary classifier</italic> which is used as a regularizer. InceptionV3, with its 42-layer-deep network, is computationally cheaper and much more efficient than other deep neural networks (<xref ref-type="bibr" rid="B28">Szegedy et&#xa0;al., 2016</xref>).</p>
</sec>
<sec id="s3_3_1_4">
<label>3.3.1.4</label>
<title>UNet with VGG19 backbone</title>
<p>The VGG model (<xref ref-type="bibr" rid="B26">Simonyan and Zisserman, 2015</xref>) derives inspiration from its predecessor, AlexNet, and is a much-improved version that uses deep convolutional neural layers to achieve better accuracy. VGG19 is the successor to the VGG16 model with 19 layers. VGG19 achieves better accuracy (<xref ref-type="bibr" rid="B25">Shu, 2019</xref>) than the VGG16 model as it can extract features better with its deep convoluted network. VGG19 has 16 convolutional layers with 3 FC layers and 5 pooling layers. Here, 2 of the 3 FC layers consist of 4096 channels each. The final FC layer originally had 1000 channels, followed by a SoftMax layer.</p>
<p>We have used the previously discussed three models as the backbone for EmergeNet, replacing the encoder part of the UNet with one of the models at a time. Owing to the symmetric structure of the UNet model, in the decoder or the expansion path, we programmatically upscale the corresponding model in a symmetric fashion to get the final output. For example, if we use VGG19 as the backbone, we are replacing the encoder part of the UNet with the VGG architecture, and in the expansion path, we are using the same VGG architecture to programmatically upscale it. These backbone models were previously trained on the significantly large well-known dataset called &#x2018;ImageNet&#x2019;, which consists of 3.2 million images (<xref ref-type="bibr" rid="B9">Deng et&#xa0;al., 2009</xref>). Thus, using these pre-trained weights allows us to benefit from transfer learning for improved accuracy and speed.</p>
</sec>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>EmergeNet loss function</title>
<p>Instead of the traditional &#x2018;binary cross-entropy&#x2019; loss (the negative average of the log of corrected predicted probabilities), EmergeNet uses a weighted sum of the two loss functions which are relevant to the task of segmentation. They are the Dice coefficient loss and focal loss. The motivation for using these two loss functions instead of cross-entropy loss is that these functions address some of the limitations of traditional cross-entropy loss. The statistical distributions of labels play a big role in training accuracy when using cross-entropy loss. The training becomes more difficult as the label distributions become more unbalanced. This is because cross-entropy loss is calculated as the average of per-pixel loss without knowing whether its adjacent pixels are boundaries or not. The Dice coefficient loss and the focal loss, discussed in detail, address these disadvantages and therefore boost the performance of the model.</p>
<p>The Dice coefficient is a statistic used to gauge the similarity of two samples and was independently developed by Thorvald S&#xf8;rensen and Lee Raymond Dice (<xref ref-type="bibr" rid="B23">S&#xf8;rensen&#x2013;Dice coefficient, 1948</xref>). It was brought to the computer vision community by (<xref ref-type="bibr" rid="B21">Milletari et&#xa0;al., 2016</xref>) for 3D medical image segmentation. The Dice loss is computed by</p>
<disp-formula>
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>*</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>*</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where, <italic>p<sub>i</sub>
</italic> and <italic>g<sub>i</sub>
</italic> represent pairs of corresponding pixel values of prediction and ground-truth, respectively. The values of <italic>p<sub>i</sub>
</italic> and <italic>g<sub>i</sub>
</italic> are either 0 or 1 in boundary detection scenarios, therefore the denominator becomes the sum of the total boundary pixels of both prediction and ground-truth and the numerator becomes the sum of correctly predicted boundary pixels because the sum increments only when <italic>p<sub>i</sub>
</italic> and <italic>g<sub>i</sub>
</italic> match (both of value 1). <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref> shows the Venn diagram for the Dice loss.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>
<bold>(A)</bold> Dice coefficient (set view); and <bold>(B)</bold> focal loss for <italic>&#x3b3;</italic>&#x2208;[0,5].</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g004.tif"/>
</fig>
<p>The Dice similarity coefficient (DSC) is a measure of the overlap between two sets (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>). In the task of boundary detection, the ground-truth boundary pixels and predicted boundary pixels can be viewed as two sets. By leveraging Dice loss, the two sets are trained to overlap gradually as training progresses. The denominator considers the total number of boundary pixels at a global scale, while the numerator considers the overlap between the two sets at a local scale. Therefore, DSC considers the loss information both locally and globally, making it a very effective loss metric for segmentation.</p>
<p>Focal loss, developed by (<xref ref-type="bibr" rid="B19">Lin et&#xa0;al., 2017</xref>), is a modified version of the Cross-Entropy (CE) loss. In the focal loss, the loss for correctly classified labels is scaled down so that the network focuses more on incorrect and low-confidence labels. In the task of segmenting a tiny foreground that relies on pixel-wise classification, a huge class imbalance occurs due to the presence of a considerably large background. Easily classified negatives comprise the majority of the loss and dominate the gradient. The focal loss is designed to address this issue by modifying the CE loss equation. The CE loss for binary classification is given as:</p>
<disp-formula>
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mtext>if&#x2009;</mml:mtext>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mtext>otherwise</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>y</italic>&#x2208;&#xb1;1 specifies the ground-truth class and <italic>p</italic>&#x2208;[0,1] is the model&#x2019;s estimated probability for the class with label <italic>y</italic>=1 .</p>
<p>(<xref ref-type="bibr" rid="B19">Lin et&#xa0;al., 2017</xref>) proposed to reshape the loss function to down-weigh easy examples and therefore focus more on training on hard negatives. Mathematically, they proposed to add a modulating factor (1&#x2212;<italic>p</italic>
<sub>
<italic>t</italic>
</sub>)<sup>
<italic>&#x3b3;</italic>
</sup> to the CE loss, with tunable focusing parameter <italic>&#x3b3;</italic>&#x2265;0 . Focal Loss is therefore defined as:</p>
<disp-formula>
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
</mml:msup>
<mml:mi>log</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>p<sub>t</sub>
</italic> is given by the equation:</p>
<disp-formula>
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mtext>if&#x2009;</mml:mtext>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mtext>otherwise</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The focal loss is visualized for several values of <italic>&#x3b3;</italic>&#x2208;[0,5] in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>.</p>
<p>From <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>, we note the following properties of the focal loss:</p>
<list list-type="bullet">
<list-item>
<p>When an example is misclassified and <italic>p</italic>
<sub>
<italic>t</italic>
</sub> is small, the modulating factor is near 1 and the loss is unaffected.</p>
</list-item>
<list-item>
<p>As <italic>p</italic>
<sub>
<italic>t</italic>
</sub>&#x2192;1 , the factor goes to 0 and the loss for well-classified examples is down-weighted.</p>
</list-item>
<list-item>
<p>The focusing parameter <italic>&#x3b3;</italic> , smoothly adjusts the rate at which easy examples are down-weighted.</p>
</list-item>
</list>
<p>The EmergeNet Loss function, <italic>loss</italic>
<sub>
<italic>e</italic>
</sub> , is calculated by the weighted sum of Dice coefficient loss, i.e., <italic>loss</italic>
<sub>
<italic>d</italic>
</sub> (defined in Eq 1), and focal loss, i.e., <italic>loss</italic>
<sub>
<italic>f</italic>
</sub> (defined in Eq 3) as follows:</p>
<disp-formula>
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>&#x3b1;</italic> is the tuning factor. For our experiment, it has been experimentally found that the optimal value of <italic>&#x3b1;</italic> is 1.</p>
</sec>
<sec id="s3_3_3">
<label>3.3.3</label>
<title>Performance-based weighted ensemble learning</title>
<p>Ensemble learning is a process by which multiple models are strategically generated and combined to solve a particular computational intelligence problem for improved performance. An ensemble model is typically constructed in two steps. First, a number of base learners are built either in parallel or in a sequence. Then, the base learners are combined using popular techniques like majority voting or weighted averaging. There are three main reasons (<xref ref-type="bibr" rid="B11">Dietterich, 1997</xref>) why the generalization ability of an ensemble is usually much stronger than that of a single learner:</p>
<list list-type="bullet">
<list-item>
<p>The training data might not provide sufficient information for choosing a single best learner. For example, many base learners could perform equally well on the training dataset. Therefore, combining these learners might be a better choice.</p>
</list-item>
<list-item>
<p>The search processes of the learning algorithms might be imperfect. For example, it might be difficult to achieve a unique best hypothesis, even if one exists, since the algorithms result in a sub-par hypothesis. This can be mitigated by the use of ensemble learning.</p>
</list-item>
<list-item>
<p>The hypothesis space being searched for might not contain the true target function, while ensembles can give some good approximation.</p>
</list-item>
</list>
<p>Instead of using state-of-the-art ensembling techniques like bagging or boosting, EmergeNet introduces a novel weighted ensembling technique that aims to calculate the weights of the individual models based on their performances. These weights are then used to reward or penalize the models. Let <italic>IoU</italic>
<sub>
<italic>i</italic>
</sub> be the Intersection over Union (IoU) score of the <italic>i<sup>th</sup>
</italic> model. We define a penalizing factor <italic>p<sub>i</sub>
</italic> as:</p>
<disp-formula>
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mo>&#x2211;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The optimal weight, <italic>w<sub>i</sub>
</italic> is then calculated as:</p>
<disp-formula>
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mo>&#x2211;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
</mml:msup>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Finally, the ensembled weighted IoU of the EmergeNet (<italic>IoU</italic>
<sub>
<italic>w</italic>
</sub> ) is computed as follows:</p>
<disp-formula>
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mo>&#x2211;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
</mml:msup>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> shows a compact view of the proposed EmergeNet architecture.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The proposed EmergeNet architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Evaluation metrics</title>
<p>The performance of our proposed EmergeNet model has been evaluated using three evaluation metrics, namely, F1-Score, Matthews Correlation Coefficient (MCC) (<xref ref-type="bibr" rid="B20">Matthews, 1975</xref>), and Intersection over Union (IoU) whereas the emergence time detection is evaluated using our proposed Emergence Time Accuracy (ETA). Accuracy (or Pixel Accuracy) is not a reliable metric for the task of segmenting tiny objects because this metric is strongly biased by classes that take a large portion of the image. Therefore, we have not used accuracy as a performance metric in this study. It is worth noting that a True Positive (TP) is an outcome where the model correctly predicts the positive class. Similarly, a True Negative (TN) is an outcome where the model correctly predicts the negative class. A False Positive (FP) is an outcome where the model incorrectly predicts the positive class and a False Negative (FN) is an outcome where the model incorrectly predicts the negative class. These metrics are defined as follows:</p>
<list list-type="bullet">
<list-item>
<p>F1-Score is the Harmonic Mean between precision and recall. The range for F1-Score is [0, 1], with 0 being the worst and 1 being the best prediction. It is governed by the equation:</p>
</list-item>
</list>
<disp-formula>
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="bullet">
<list-item>
<p>MCC is an improved metric which takes into account true and false positives and negatives and is generally regarded as a balanced measure that can be used even if the classes are of very different sizes. It has a range of -1 to 1 where -1 is a completely negative correlation between ground-truth and predicted value whereas +1 indicates a completely positive correlation between the ground-truth and predicted value.</p>
</list-item>
</list>
<disp-formula>
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mtext>MCC</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>&#xa0;TP</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<list list-type="bullet">
<list-item>
<p>Intersection over Union (IoU) is a number from 0 to 1 that specifies the amount of overlap between the prediction and ground-truth.</p>
</list-item>
</list>
<disp-formula>
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The emergence time is defined as the timestamp of the image in which EmergeNet first detects the coleoptile(s).</p>
<p>Let <italic>M</italic>={<italic>&#x3b1;</italic>
<sub>1</sub>,<italic>&#x3b1;</italic>
<sub>2</sub>,&#x2026;,<italic>&#x3b1;</italic>
<sub>
<italic>n</italic>
</sub>} , where <italic>&#x3b1;<sub>i</sub>
</italic> denotes the image for a seeded pot obtained at timestamp <italic>t<sub>i</sub>
</italic>, <italic>n</italic> denotes the total number of images in the sequence, where <italic>t</italic>
<sub>
<italic>i</italic>
</sub>&#xa0;&lt;&#xa0;<italic>t</italic>
<sub>
<italic>i</italic>+1</sub> , &#x2200;&#xa0;1&#x2264;<italic>i</italic>&lt;<italic>n</italic> . The emergence time for a pot is given by the first timestamp EmergeNet finds the coleoptile. Thus,</p>
<disp-formula>
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2260;</mml:mo>
<mml:mo>&#x2205;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#x2205;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The emergence time accuracy (ETA) is determined by comparing the time computed from the results from EmergeNet with the ground-truth, obtained by careful manual inspection of the image sequence.</p>
<p>Given an image sequence, the detection of the emergence time is considered accurate if the time predicted based on the results of EmergeNet (Eq 12) matches the ground-truth, i.e.,</p>
<disp-formula>
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>G</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>GroundTruth</italic>(<italic>M</italic>) is the timestamp of the emergence determined manually.</p>
<p>ETA is given by the proportion of emergences accurately identified by EmergeNet. Thus,</p>
<disp-formula>
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>A</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental analysis</title>
<p>This section discusses the experimental setup, the benchmark dataset, the evaluation metrics used to evaluate the proposed method, and the results obtained from our experiments.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental setup</title>
<p>The experimental analyses are performed using the Kaggle Notebook, a cloud computational environment that provides a free platform to run code in Python using dedicated GPUs. Kaggle Notebooks run in a remote computational environment and each Notebook editing session is provided with many resources. We used a GPU Kernel with Tesla P100 16 GB VRAM as GPU, with 13 GB RAM along with a 2-core of Intel Xeon as CPU. The training masks are generated using the VGG annotator tool. Python is featured with a plethora of useful packages, like, OpenCV, TensorFlow, Keras, Scikit-learn, etc., which are used to train the model and evaluate its performance. The number of images used for training was 260 (130 images and their corresponding masks). The execution time for training the EmergeNet was 1.5 hours. Compared to other deep neural networks, EmergeNet took less time to train as it benefits from transfer learning. We trained each model until the IoU curve for each of them reached saturation, and no further improvement was possible.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Results</title>
<p>We present our results in two different parts. First, we present a comparative study of EmergeNet and UNet in terms of performance. We have also compared the performance of all three individual backbone networks with EmergeNet. A more detailed analysis of the performance of EmergeNet under dark lighting conditions is also presented. In the second part, we analyzed the growth monitoring of maize coleoptiles as well as their emergence timing detection.</p>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>Comparative study</title>
<p>
<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> shows the comparison of masks generated by UNet and EmergeNet using a test image sequence from the UNL-MED. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref> shows the masks generated by the standard UNet model and their corresponding ground-truth. Note that the generated masks do not accurately match with the ground-truth. Furthermore, the UNet model failed to detect the emergence of coleoptiles in several cases. This is generally the case with generic models which are not sophisticated enough. The IoU obtained by the standard UNet model for this test sample is 69%. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref> shows the masks generated by the proposed EmergeNet model and their corresponding ground-truth.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Illustration of segmentation performance on a test image from UNL-MED by <bold>(A)</bold> UNet and <bold>(B)</bold> EmergeNet.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g006.tif"/>
</fig>
<p>In contrast to the UNet model&#x2019;s results, the masks generated using EmergeNet closely correspond to the ground-truth. EmergeNet neither incorrectly generated a mask (when there was no emergence), nor failed to produce a mask when there was a coleoptile. The overall IoU of EmergeNet for this test sample is 99.40%, a significant improvement over UNet. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> shows the comparative analysis of UNet and EmergeNet in terms of the three evaluation metrics, namely, F1-Score, MCC, and IoU, for all images of UNL-MED. It is evident from the table that EmergeNet significantly outperforms the UNet model.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Results of comparison between UNet and EmergeNet on UNL-MED under all lighting conditions in terms of F1-Score, MCC, and IoU.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="left">F1-Score</th>
<th valign="top" align="center">MCC</th>
<th valign="top" align="center">IoU</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">UNet</td>
<td valign="top" align="left">0.6813</td>
<td valign="top" align="center">0.7040</td>
<td valign="top" align="center">0.7690</td>
</tr>
<tr>
<td valign="top" align="left">EmergeNet</td>
<td valign="top" align="left">0.9470</td>
<td valign="top" align="center">0.9894</td>
<td valign="top" align="center">0.9820</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>While a confusion matrix (CM) is a powerful visualization technique to summarize the performance of a supervised classification task, it does not provide valuable insights into the model&#x2019;s performance for image segmentation since the data is highly imbalanced toward the background class. The normalized CMs for a random test image for the standard UNet and EmergeNet are shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. It is evident from the figure that the background is significantly larger than the maize coleoptile. A more accurate representation of the classifier&#x2019;s performance can be derived by overlaying the values of the confusion matrix on the coleoptile mask generated by the classifier. <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7A</bold>
</xref> shows that majority of the mask generated by the standard UNet (shown in magenta) is incorrectly labeled as the coleoptile, i.e., false positive. Only a very small portion of the mask is accurately labeled (shown in cyan), denoting the true positives. <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7B</bold>
</xref> shows the mask overlaid by the values of the corresponding confusion matrix for EmergeNet. It shows that a significant majority of the mask is correctly labeled (shown in cyan), and only a few pixels, mostly along the border, are false positives (shown in magenta). There are no false negatives or true negatives for EmergeNet. This demonstrates the efficacy of EmergeNet and its superiority over the standard UNet for this application.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>
<bold>(A)</bold> Juxtaposition of confusion matrix (left) and its corresponding overlay mask from standard UNet (right). <bold>(B)</bold> Juxtaposition of confusion matrix (left) and its corresponding overlay mask from the EmergeNet (right).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g007.tif"/>
</fig>
<p>To further add credibility to the accuracy and robustness of the integrated network, i.e., EmergeNet, we performed a comparative study among the individual networks with EmergeNet. In each case, the masks generated by EmergeNet are better than that created by the individual networks. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> compares the masks generated by UNet with SEResNet as its backbone, and EmergeNet. The IoU of EmergeNet is higher by 2.21%. InceptionV3 is a very powerful network on its own, and therefore, the UNet structure with InceptionV3 as its backbone is expected to perform remarkably well. Such is the case as depicted in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, however, EmergeNet still beats the IoU score by 0.11% which is impressive considering the fact that it becomes exponentially more difficult to improve the results above a certain threshold value. Finally, EmergeNet beats UNet with VGG as its backbone by 0.54% in terms of IoU metrics as shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>. Conclusively, it can be inferred that the integrated structure of EmergeNet plays a significant role in bringing the best of all the individual networks and performs better than all of them, thereby proving its worth as a segmentation model.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Illustration of segmentation performance on a test image from UNL-MED by <bold>(A)</bold> SEResNet and <bold>(B)</bold> EmergeNet.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g008.tif"/>
</fig>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Illustration of segmentation performance on a test image from UNL-MED by <bold>(A)</bold> InceptionV3 and <bold>(B)</bold> EmergeNet.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g009.tif"/>
</fig>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Illustration of segmentation performance on a test image from UNL-MED by <bold>(A)</bold> VGG19 and <bold>(B)</bold> EmergeNet.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g010.tif"/>
</fig>
<p>To demonstrate the efficacy of EmergeNet under extremely low light conditions, we conducted the same experimental analyses by considering the images of UNL-MED that were captured in a dark environment only. <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> summarizes the results of the comparison between the UNet and EmergeNet in dark conditions. Results show that EmergeNet significantly outperformed the standard UNet along all three evaluation metrics.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Results of comparison between UNet and EmergeNet on UNL-MED only under dark environmental conditions in terms of F1-Score, MCC, and IoU.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">F1-Score</th>
<th valign="top" align="center">MCC</th>
<th valign="top" align="center">IoU</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">UNet</td>
<td valign="top" align="left">0.6065</td>
<td valign="top" align="center">0.6000</td>
<td valign="top" align="center">0.7774</td>
</tr>
<tr>
<td valign="top" align="left">EmergeNet</td>
<td valign="top" align="left">0.9709</td>
<td valign="top" align="center">0.9709</td>
<td valign="top" align="center">0.9735</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>Growth monitoring</title>
<p>Thus, EmergeNet can efficiently monitor the growth of maize coleoptiles even at extremely low light conditions. We defined a new measure called the ETA to evaluate the accuracy of emergence in Section 4.4. <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> displays a sequence of images that show the emergence and growth of coleoptiles computed by EmergeNet. It shows that EmergeNet detected the emergence of coleoptile with 100% ETA. Furthermore, it correctly tracks the growth of the coleoptile, as demonstrated by the increasing size of the generated masks in the time-lapse imagery. Note that the first two subfigures, i.e., <xref ref-type="fig" rid="f11">
<bold>Figures&#xa0;11A, B</bold>
</xref>, do not contain any masks because the emergence has not taken place yet. Out of nine maize seeds sown in the nine pots, only four of them emerged earlier, as shown by the tiny masks in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11C</bold>
</xref>. Subsequently two more coleoptiles emerged, one in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11D</bold>
</xref> and the other in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11G</bold>
</xref>. Three seeds failed to emerge in the experiment. After their emergence, all six coleoptiles are correctly tracked. Thus, EmergeNet accurately identifies the emergence events and tracks the growth of the coleoptiles over time.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>
<bold>(A&#x2013;L)</bold> Illustration of emergence timing detection and growth monitoring of a maize coleoptile in a time-lapse test image sequence of UNL-MED.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g011.tif"/>
</fig>
<p>We conducted an experiment to demonstrate the accuracy of the coleoptile size measured by the total number of constituent pixels of the generated masks by comparing them with the ground-truth. The result of the comparison is shown in <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>. The figure shows the coleoptile size of the generated mask (shown in blue) significantly overlaps with the coleoptile size of the ground-truth (shown in red). Thus, EmergeNet not only produces high-precision masks but also helps in the growth monitoring of coleoptiles with very high accuracy. <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> shows the Pearson correlation coefficient between the coleoptile size of the mask generated by EmergeNet and the ground-truth. Pearson correlation coefficient is a measure of the linear relationship between two variables. It has a value between -1 to 1, with a value of -1 denoting a total negative linear correlation, 0 being no correlation, and + 1 denoting a total positive correlation. The table shows a high positive correlation between the ground-truth mask and the generated mask in terms of coleoptile size.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Coleoptile size measured by the total number of constituent pixels of the ground-truth (shown in blue) and the generated mask (shown in red) as a function of time.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1084778-g012.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Pearson correlation table between ground-truth and generated mask.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">Ground-truth</th>
<th valign="top" align="center">Generated mask</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Ground-truth</td>
<td valign="top" align="left">1.000000</td>
<td valign="top" align="left">0.999412</td>
</tr>
<tr>
<td valign="top" align="left">Generated mask</td>
<td valign="top" align="left">0.999412</td>
<td valign="top" align="left">1.000000</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s5" sec-type="discussion">
<label>5</label>
<title>Discussion</title>
<p>The timing of germination is a paramount physiological factor for seed quality determination that encompasses a set of broad concerns, including vigor, dormancy mechanisms, pests, pathogens, genetic integrity, cost of establishment, field maintenance to prevent contamination with weeds or unwanted seed, and isolation distances to prevent cross-pollination <xref ref-type="bibr" rid="B27">Stiller et&#xa0;al. (2010)</xref>. Thus, research attention for automated emergence timing determination based on computer vision and artificial intelligence techniques to replace tedious manual human labor is more crucial than ever. In this paper, we proposed an ensemble deep-learning based segmentation model based on the UNet architecture for coleoptile emergence time detection. The proposed EmergeNet outperforms the UNet by a significant margin as demonstrated by the experimental analyses in Section 4.2. The success of EmergeNet is attributed to successful base backbone architectures, customized loss function, and a novel penalizing factor in the ensemble technique. The accuracy of any image-based phenotypes depends on the accuracy of the underlying segmentation model <xref ref-type="bibr" rid="B5">Das Choudhury (2020)</xref>. Thus, the ensemble segmentation model introduced in this paper has the potential to be extended to other automated phenotyping applications.</p>
<p>The proposed EmergeNet model significantly outperforms the widely used UNet architecture. The IoU metric for mask generation for EmergeNet is 99.4%; in comparison, the standard UNet has an IoU of 69% (see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). By combining the three UNet architectures with powerful backbones, along with the proposed custom loss function and a novel ensemble technique, EmergeNet reduces the number of false positives (see <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7B</bold>
</xref>). One of the key contributions of this model is its success in extremely low light, as evident from <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. Overall, EmergeNet is able to detect the emergence timing of the maize coleoptile with 100% accuracy. Furthermore, EmergeNet can accurately monitor the growth of the coleoptiles over time, as demonstrated by a very high correlation (0.999) between the generated masks and the ground-truth.</p>
</sec>
<sec id="s6" sec-type="conclusion">
<label>6</label>
<title>Conclusion</title>
<p>The timing of important events in a plant&#x2019;s life, for instance, germination, the emergence of a new leaf, flowering, fruiting, and onset of senescence, is crucial in the understanding of the overall plant&#x2019;s vigor, which is likely to vary with the interaction between genotype and environment, and are referred to as event-based phenotypes. This paper introduces a novel deep-learning model called EmergeNet to detect the timing of the emergence of a maize seedling and track its growth over a time-lapse video sequence. EmergeNet is based on an ensemble model that integrates SEResNet18, InceptionV3, and VGG19, such that it overcomes the challenge of detecting a tiny living object and tracks its changes in shape and appearance in the presence of cluttered soil background and extreme variation of illuminations. Furthermore, the paper introduces a benchmark dataset called UNL-MED. Experimental evaluation on UNL-MED shows the capability of EmergeNet to detect the timing of emergence with 100% accuracy as compared with human-perceived ground-truth. It is also experimentally demonstrated that EmergeNet significantly outperforms its base model UNet in the task of segmentation. EmergeNet incorporates three pre-trained networks including all their weights, and hence, it requires high-end computing power for efficient training. Additionally, EmergeNet is trained on only one type of plant constrained by a set of external environmental conditions. Future work will consider the detection of multiple coleoptiles with or without the presence of weeds in the same pot.</p>
</sec>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <uri xlink:href="https://plantvision.unl.edu/dataset">https://plantvision.unl.edu/dataset</uri>.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>AD developed and implemented the algorithm, conducted experimental analysis, and contributed to manuscript writing. SD conceived the idea, led the dataset design, conducted experimental analysis, led the manuscript writing, and supervised the research. AS, AKD, and TA critically reviewed the manuscript and provided constructive feedback throughout the process. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>This work is supported by Agricultural Genome to Phenome Initiative Seed Grant [grant no. 2021-70412-35233] from the USDA National Institute of Food and Agriculture.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>Authors are thankful to Dr. Vincent Stoerger, the Plant Phenomics Operations Manager at the University of Nebraska-Lincoln, USA, for his support in setting up experiments to create the dataset used in this study.</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Agarwal</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Detection of Plant Emergence Based on Spatio Temporal Image Sequence Analysis</article-title>. <source>Master&#x2019;s thesis</source> (<publisher-name>The University of Nebraska-Lincoln</publisher-name>).</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aggarwal</surname> <given-names>J. K.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Human motion analysis: A review</article-title>. <source>Comput. Vision image understanding</source> <volume>73</volume>, <fpage>428</fpage>&#x2013;<lpage>440</lpage>. doi: <pub-id pub-id-type="doi">10.1006/cviu.1998.0744</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arulmurugan</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hema Rajini</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Pradeep</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Ensemble of deep learning based segmentation with classification model for skin lesion diagnosis using dermoscopic images</article-title>. <source>J. Comput. Theor. Nanoscience</source> <volume>18</volume>, <fpage>710</fpage>&#x2013;<lpage>721</lpage>. doi: <pub-id pub-id-type="doi">10.1166/jctn.2021.9667</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bashyam</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Das Choudhury</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Samal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Awada</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Visual growth tracking for automated leaf stage monitoring based on image sequence analysis</article-title>. <source>Remote Sens.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs13050961</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Das Choudhury</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Segmentation techniques and challenges in plant phenotyping</article-title>,&#x201d; in <source>Intelligent image analysis for plant phenotyping</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Samal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Das Choudhury</surname> <given-names>S.</given-names>
</name>
</person-group> (<publisher-loc>Boca Raton, Florida</publisher-loc>: <publisher-name>CRC Press, Taylor &amp; Francis Group</publisher-name>), <fpage>69</fpage>&#x2013;<lpage>91</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Das Choudhury</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bashyam</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Samal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Awada</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Holistic and component plant phenotyping using temporal image sequence</article-title>. <source>Plant Methods</source> <volume>14</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-018-0303-x</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Das Choudhury</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Samal</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Structural high-throughput plant phenotyping based on image sequence analysis</article-title>,&#x201d; in <source>Intelligent image analysis for plant phenotyping</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Samal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Das Choudhury</surname> <given-names>S.</given-names>
</name>
</person-group> (<publisher-loc>Boca Raton, Florida</publisher-loc>: <publisher-name>CRC Press, Taylor &amp; Francis Group</publisher-name>), <fpage>93</fpage>&#x2013;<lpage>117</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Das Choudhury</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Samal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Awada</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Leveraging image analysis for high-throughput plant phenotyping</article-title>. <source>Front. Plant Sci.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2019.00508</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Deng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Socher</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L.-J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Fei-Fei</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Imagenet: A large-scale hierarchical image database</article-title>,&#x201d; in <source>2009 IEEE conference on computer vision and pattern recognition (Ieee)</source> (<publisher-loc>Miami, FL, USA</publisher-loc>), <fpage>248</fpage>&#x2013;<lpage>255</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dhaka</surname> <given-names>V. S.</given-names>
</name>
<name>
<surname>Meena</surname> <given-names>S. V.</given-names>
</name>
<name>
<surname>Rani</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Sinwar</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Ijaz</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Wo&#x17a;niak</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A survey of deep convolutional neural networks applied for prediction of plant leaf diseases</article-title>. <source>Sensors</source> <volume>21</volume>, <fpage>4749</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s21144749</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dietterich</surname> <given-names>T. G.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Machine-learning research</article-title>. <source>AI magazine</source> <volume>18</volume>, <fpage>97</fpage>&#x2013;<lpage>97</lpage>. doi: <pub-id pub-id-type="doi">10.1609/aimag.v18i4.1324</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wojek</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Schiele</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Perona</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Pedestrian detection: A benchmark</article-title>,&#x201d; in <source>2009 IEEE conference on computer vision and pattern recognition (IEEE)</source> (<publisher-loc>Miami, FL, USA</publisher-loc>), <fpage>304</fpage>&#x2013;<lpage>311</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Dutta</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zissermann</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>) <source>VGG image annotator (VIA)</source>. Available at: <uri xlink:href="http://www.robots.ox.ac.uk/vgg/software/via/">http://www.robots.ox.ac.uk/vgg/software/via/</uri>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Golla</surname> <given-names>A.-K.</given-names>
</name>
<name>
<surname>Bauer</surname> <given-names>D. F.</given-names>
</name>
<name>
<surname>Schmidt</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Russ</surname> <given-names>T.</given-names>
</name>
<name>
<surname>N&#xf6;renberg</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Chung</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Convolutional neural network ensemble segmentation with ratio-based sampling for the arteries and veins in abdominal ct scans</article-title>. <source>IEEE Trans. Biomed. Eng.</source> <volume>68</volume>, <fpage>1518</fpage>&#x2013;<lpage>1526</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TBME.2020.3042640</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Squeeze-and-excitation networks</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source> (<publisher-loc>Salt Lake City, UT, USA</publisher-loc>), <fpage>7132</fpage>&#x2013;<lpage>7141</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>Y.-W.</given-names>
</name>
<name>
<surname>Byun</surname> <given-names>Y.-C.</given-names>
</name>
<name>
<surname>Krishna</surname> <given-names>A. V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Portrait segmentation using ensemble of heterogeneous deep-learning models</article-title>. <source>Entropy</source> <volume>23</volume>, <fpage>197</fpage>. doi: <pub-id pub-id-type="doi">10.3390/e23020197</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lagree</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Mohebpour</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Meti</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Saednia</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>F.-I.</given-names>
</name>
<name>
<surname>Slodkowska</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>A review and comparison of breast tumor cell nuclei segmentation performances using deep convolutional neural networks</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-021-87496-1</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Mitra</surname> <given-names>N. J.</given-names>
</name>
<name>
<surname>Chamovitz</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Cohen-Or</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Analyzing growing plants from 4d point cloud data</article-title>. <source>ACM Trans. Graphics</source> <volume>32</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/2508363.2508368</pub-id>.Yangyan2013</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>T.-Y.</given-names>
</name>
<name>
<surname>Goyal</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Focal loss for dense object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE international conference on computer vision</source> (<publisher-loc>Venice, Italy</publisher-loc>), <fpage>2980</fpage>&#x2013;<lpage>2988</lpage>. lin2017focal.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matthews</surname> <given-names>B. W.</given-names>
</name>
</person-group> (<year>1975</year>). <article-title>Comparison of the predicted and observed secondary structure of t4 phage lysozyme</article-title>. <source>Biochim. Biophys. Acta (BBA)-Protein Structure</source> <volume>405</volume>, <fpage>442</fpage>&#x2013;<lpage>451</lpage>. matthews1975comparison. doi: <pub-id pub-id-type="doi">10.1016/0005-2795(75)90109-9</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Milletari</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Navab</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Ahmadi</surname> <given-names>S.-A.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>V-Net: Fully convolutional neural networks for volumetric medical image segmentation</article-title>,&#x201d; in <source>2016 fourth international conference on 3D vision (3DV) (IEEE)</source> (<publisher-loc>Stanford, CA, USA</publisher-loc>), <fpage>565</fpage>&#x2013;<lpage>571</lpage>. milletari2016v.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Min</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A new approach to track multiple vehicles with the combination of robust detection and two classifiers</article-title>. <source>IEEE Trans. Intelligent Transportation Syst.</source> <volume>19</volume>, <fpage>174</fpage>&#x2013;<lpage>186</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TITS.2017.2756989</pub-id>Min2018</citation>
</ref>
<ref id="B23">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>S&#xf8;rensen&#x2013;Dice coefficient</collab>
</person-group> (<year>1948</year>) <source>S&#xf8;rensen&#x2013;dice coefficient &#x2014; Wikipedia, the free encyclopedia</source> (Accessed <access-date>23-July-2021</access-date>). enwiki:1023592604.</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Scharr</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Bruns</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Fischbach</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Roussel</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Scholtes</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Stein</surname> <given-names>J. v.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Germination detection of seedlings in soil: A system, dataset and challenge</article-title>,&#x201d; in <source>Computer vision - ECCV 2020 workshops</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Bartoli</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Fusiello</surname> <given-names>A.</given-names>
</name>
</person-group> (<publisher-name>Springer International Publishing</publisher-name>), <fpage>360</fpage>&#x2013;<lpage>374</lpage>. Hanno2020.</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shu</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Deep learning for image classification on very small datasets using transfer learning shu</source> (<publisher-loc>MS thesis, Iowa State University, USA</publisher-loc>).</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). <source>Very deep convolutional networks for large-scale image recognition</source> (<publisher-loc>San Diego, CA, USA</publisher-loc>:<publisher-name>International Conference on Learning Representations</publisher-name>), arXiv preprint arXiv:1409.1556 vgg.</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Stiller</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Bocek</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Hecht</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Machado</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Racz</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Waldburger</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2010</year>). <source>Mobile systems IV</source> (<publisher-loc>Zurich, Switzerland</publisher-loc>:<publisher-name>Tech. rep., University of Zurich, Department of Informatics instance</publisher-name>), <fpage>1290</fpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Vanhoucke</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Ioffe</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Shlens</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wojna</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Rethinking the inception architecture for computer vision</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source> (<publisher-loc>Las Vegas, Nevada, USA</publisher-loc>), <fpage>2818</fpage>&#x2013;<lpage>2826</lpage>. [Dataset].</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Walsh</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Koirala</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Mango fruit load estimation using a video based mangoyolo&#x2013;kalman filter&#x2013;hungarian algorithm method</article-title>. <source>Sensors</source> <volume>19</volume>, <fpage>2742</fpage>. Zhenglin2019. doi: <pub-id pub-id-type="doi">10.3390/s19122742</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xing</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ai</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Lao</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Multiple player tracking in sports video: A dual-mode two-way bayesian inference approach with progressive observation modeling</article-title>. <source>IEEE Trans. Image Process.</source> <volume>20</volume>, <fpage>1652</fpage>&#x2013;<lpage>1667</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2010.2102045</pub-id>. Xing2011.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep learning for multiple object tracking: A survey</article-title>. <source>IET Comput. Vision</source> <volume>13</volume>, <fpage>355</fpage>&#x2013;<lpage>368</lpage>. xu2019deep. doi: <pub-id pub-id-type="doi">10.1049/iet-cvi.2018.5598</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yilmaz</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Javed</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Object tracking: A survey</article-title>. <source>ACM computing surveys (CSUR</source> <volume>38</volume>, <fpage>13</fpage>&#x2013;<lpage>es</lpage>. yilmaz2006object. doi: <pub-id pub-id-type="doi">10.1145/1177352.1177355</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>