<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Neurosci.</journal-id>
<journal-title>Frontiers in Computational Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-5188</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fncom.2022.842760</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>U-RISC: An Annotated Ultra-High-Resolution Electron Microscopy Dataset Challenging the Existing Deep Learning Algorithms</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Shi</surname> <given-names>Ruohua</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1593315/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Wenyao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1439745/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Zhixuan</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1721431/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>He</surname> <given-names>Liuyuan</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/276557/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sheng</surname> <given-names>Kaiwen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1445453/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ma</surname> <given-names>Lei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1445881/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Du</surname> <given-names>Kai</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1445441/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Jiang</surname> <given-names>Tingting</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1600994/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Huang</surname> <given-names>Tiejun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1598097/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Beijing Academy of Artificial Intelligence</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>National Engineering Research Center of Visual Technology, School of Computer Science, Peking University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Institute for Artificial Intelligence, Peking University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Youhui Zhang, Tsinghua University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Xuejin Chen, University of Science and Technology of China, China; Guozhang Chen, Graz University of Technology, Austria</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Kai Du <email>kai.du&#x00040;pku.edu.cn</email></corresp>
<corresp id="c002">Tingting Jiang <email>ttjiang&#x00040;pku.edu.cn</email></corresp>
<fn fn-type="equal" id="fn001"><p>&#x02020;These authors have contributed equally to this work and share first authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>04</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>16</volume>
<elocation-id>842760</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>12</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2022 Shi, Wang, Li, He, Sheng, Ma, Du, Jiang and Huang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Shi, Wang, Li, He, Sheng, Ma, Du, Jiang and Huang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Connectomics is a developing field aiming at reconstructing the connection of the neural system at the nanometer scale. Computer vision technology, especially deep learning methods used in image processing, has promoted connectomic data analysis to a new era. However, the performance of the state-of-the-art (SOTA) methods still falls behind the demand of scientific research. Inspired by the success of ImageNet, we present an annotated ultra-high resolution image segmentation dataset for cell membrane (U-RISC), which is the largest cell membrane-annotated electron microscopy (EM) dataset with a resolution of 2.18 nm/pixel. Multiple iterative annotations ensured the quality of the dataset. Through an open competition, we reveal that the performance of current deep learning methods still has a considerable gap from the human level, different from ISBI 2012, on which the performance of deep learning is closer to the human level. To explore the causes of this discrepancy, we analyze the neural networks with a visualization method, which is an attribution analysis. We find that the U-RISC requires a larger area around a pixel to predict whether the pixel belongs to the cell membrane or not. Finally, we integrate the currently available methods to provide a new benchmark (0.67, 10% higher than the leader of the competition, 0.61) for cell membrane segmentation on the U-RISC and propose some suggestions in developing deep learning algorithms. The U-RISC dataset and the deep learning codes used in this study are publicly available.</p></abstract>
<kwd-group>
<kwd>connectomics</kwd>
<kwd>EM dataset</kwd>
<kwd>deep learning</kwd>
<kwd>automatic cell segmentation</kwd>
<kwd>transfer learning</kwd>
</kwd-group>
<contract-num rid="cn001">62088102</contract-num>
<contract-sponsor id="cn001">Natural Science Foundation of Beijing Municipality<named-content content-type="fundref-id">10.13039/501100004826</named-content></contract-sponsor>
<counts>
<fig-count count="9"/>
<table-count count="3"/>
<equation-count count="11"/>
<ref-count count="61"/>
<page-count count="15"/>
<word-count count="9525"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Accurate descriptions of neurons and their connections are fundamental to modern neuroscience. By depicting neurons with the help of the Golgi-staining method (Golgi, <xref ref-type="bibr" rid="B20">1885</xref>), Cajal proposed the classic &#x0201C;Neuron Doctrine&#x0201D; more than a century ago (y Cajal, <xref ref-type="bibr" rid="B57">1888</xref>), which opened a new era in modern neuroscience. Nowadays, the development of electron microscopy (EM) has enabled us to further explore the structural details of the neural system at nanometer (nm) scales (Shawn, <xref ref-type="bibr" rid="B44">2016</xref>; Kornfeld and Denk, <xref ref-type="bibr" rid="B27">2018</xref>), opening up a new field called, &#x0201C;Connectomics&#x0201D; that aims to reconstruct every single connection in the neural system. One milestone of Connectomics is the <italic>Caenorhabditis elegans</italic> project (White et al., <xref ref-type="bibr" rid="B55">1986</xref>) which maps all 302 neurons and 7,000 connections in a worm. Recently, a small piece of the human cortex was imaged with a high-speed scanning EM, which maps &#x0007E;50,000 neurons and 110,000,000 synaptic connections (Shapson-Coe et al., <xref ref-type="bibr" rid="B43">2021</xref>). Connectomic data increase exponentially with a higher resolution of EM and a larger neural tissue volume, even reaching the petabyte (PB) scale (Shapson-Coe et al., <xref ref-type="bibr" rid="B43">2021</xref>). Just as it took almost 15 years to complete the connectome of <italic>C.elegans</italic>, the structural reconstruction for higher-level creatures is becoming more and more daunting with the explosion of connectomic data. Among many bottlenecks, accurate annotation from large amounts of EM images is the first one that has to be solved.</p>
<p>Manual annotation of all the connectomic data is infeasible because of the high annotation cost. To reduce the burden of manual annotation for humans, one would hope to enable a machine to annotate the connectomic data with near-human performance automatically. Hopes are higher today because of the rapid development of deep learning methods. However, even with deep learning, it still requires tremendous efforts to achieve human-level performance on this challenging task. There were a few successful experiences to learn from the computer science community to make the deep learning method fully comparable to humans in Connectomics. The success of deep learning methods highly depends on the amount of training data and the quality of annotation. For example, in the task of image classification, ImageNet (Russakovsky et al., <xref ref-type="bibr" rid="B39">2015</xref>) has set up a research paradigm in applying deep learning methods for vision tasks. In 2009, by releasing a large-scale accurately annotated dataset, ImageNet provided a benchmark (72%) for image classification. From 2010 to 2017, a challenge called, &#x0201C;The ImageNet Large Scale Visual Recognition Challenge (ILSVRC)&#x0201D; was organized every year. This challenge significantly boosted the development of deep learning algorithms. Many champions of this challenge have become the milestones for deep learning methods, such as AlexNet (Krizhevsky et al., <xref ref-type="bibr" rid="B28">2012</xref>), VGG (Simonyan and Zisserman, <xref ref-type="bibr" rid="B45">2014</xref>), GoogleNet (Szegedy et al., <xref ref-type="bibr" rid="B48">2015</xref>), and ResNet (He et al., <xref ref-type="bibr" rid="B23">2016</xref>). As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, deep learning performance on image classification finally exceeded the human level (95%) after 8 years of development. To summarize, there is a roadmap for the success of ImageNet, which includes three key steps: the first step is to establish a large-scale dataset with high-quality annotation, which is very important for deep learning. Based on the dataset, the second step organizes a challenge that can evaluate algorithms at a large scale and allow researchers to estimate the progress of their algorithms, taking advantage of the expensive annotation effort. The third step is the design of new algorithms based on the previous two steps. Each of the three stages is indispensable.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The history of ImageNet. The blue and red dot-dash lines represent the human performance of ImageNet classification and U-RISC segmentation, respectively. The red circle shows the current benchmark performance of U-RISC segmentation, and the red dot line shows the expected improvement of deep learning methods.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0001.tif"/>
</fig>
<p>Following the success of ImageNet, significant progress in the automatic segmentation of EM was achieved by the 2012 IEEE International Symposium on Biomedical Imaging (ISBI 2012), which was the first challenge on the automatic segmentation of EM in releasing a publicly available dataset (Arganda-Carreras et al., <xref ref-type="bibr" rid="B7">2015</xref>). The state-of-the-art (SOTA) methods exhibited an unprecedented accuracy in EM cellular segmentation on the dataset of ISBI 2012. In particular, the deep learning method, &#x0201C;U-Net,&#x0201D; (Ronneberger et al., <xref ref-type="bibr" rid="B37">2015</xref>) which was first proposed during the challenge, becomes the backbone of many SOTA methods in the field. However, today many deep learning methods have become &#x0201C;exceedingly accurate,&#x0201D; and are likely to be saturated at the ISBI 2012 (Arganda-Carreras et al., <xref ref-type="bibr" rid="B7">2015</xref>). In addition, ISBI 2012 images are 512 &#x000D7; 512 pixels with a resolution of 4 &#x000D7; 4 nm/pixels, while there are many EM images with higher resolution in connectomics because enough high resolution is essential to unravel the neural structures unambiguously. For instance, 2 nm has been suggested as the historical &#x0201C;gold standard&#x0201D; to identify synapses (DeBello et al., <xref ref-type="bibr" rid="B16">2014</xref>), in particular, to identify gap junctions (Leitch, <xref ref-type="bibr" rid="B29">1992</xref>), which are common in the neural tissues (Anderson et al., <xref ref-type="bibr" rid="B4">2009</xref>). It is not clear if previous classic deep learning methods developed on the EM images with relatively lower resolution can still work well on datasets with higher resolutions.</p>
<p>Here, to promote the deep learning algorithms in EM datasets, we initiated a new roadmap: We first annotated the retinal connectomic data, RC1, from rabbit (Anderson et al., <xref ref-type="bibr" rid="B5">2011</xref>) and presented a brand new annotated EM dataset named, ultra-high resolution image segmentation dataset for cell membrane (U-RISC). Compared to ISBI 2012, the U-RISC has a higher resolution of 2.18 nm/pixel and a larger size of 9,958 &#x000D7; 9,959 pixels. The precision of the annotation was ensured by multi-steps of iterative verification, costing over 10,000 labor hours in total. Next, based on the U-RISC, a competition of cellular membrane prediction was also organized. Surprisingly, from 448 domestic participants/teams, it was observed that the top performance of deep learning methods on the U-RISC (&#x0007E;0.6, F1-score) was far below the human-level accuracy (&#x0003E;0.9), in contrast to the near-human performance of deep learning methods in ISBI 2012. We then made fair comparisons between ISBI 2012 and U-RISC with the same segmentation methods, including U-Net. The comparison results confirmed that U-RISC indeed provides new challenges to the existing deep learning methods. The U-Net, for example, dropped from 0.97 in ISBI 2012 to 0.57 in the U-RISC. To further explore how these methods work on segmentation tasks, we introduced a gradient-based attribution method, an integrated gradient (IG; Sundararajan et al., <xref ref-type="bibr" rid="B47">2017</xref>), to analyze ISBI 2012 and the U-RISC. The result showed that when deciding on whether a pixel belonged to a cell membrane or not, deep learning methods represented by the U-Net would refer to a larger attribution region on the U-RISC (about four times on average) than that on ISBI 2012. This suggests that the deep learning methods might require more background information to decide the segmentation of the U-RISC dataset. Finally, we integrated the currently available advanced methods, combining the U-Net and transferring the learning recently introduced (Conrad and Narayan, <xref ref-type="bibr" rid="B14">2021</xref>), and provided a benchmark (0.6659), which is about 10% higher than the leader board (0.6070), for the U-RISC.</p>
<p>Overall, our contribution in this study lies mainly in the following three parts: (1) we provided the community with a brand new publicly available annotated mammalian EM dataset with the highest known resolution (&#x0007E;2.18 nm/pixel) and the largest image size (9,958 &#x000D7; 9,959 pixels); (2) we organized a competition and made a comprehensive analysis to reveal the challenges of U-RISC in the deep learning methods; (3) we improved the benchmark with 10% to the F1-score of 0.6659. In the Discussion, we proposed further suggestions for improving the segmentation methods from the perspectives of model design, loss function design, data processing, etc. We hope our dataset and analysis can help researchers gain insights into designing more robust methods, which can finally accelerate the speed of untangling brain connectivity.</p>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and Methods</title>
<sec>
<title>Datasets</title>
<p>The U-RISC dataset was annotated upon RC1, a large-scale retinal serial section transmission electron microscopic (ssTEM) dataset, publicly available upon request and described in detail in the study of Anderson et al. (<xref ref-type="bibr" rid="B5">2011</xref>). The RC1 came from the retina of a light-adapted female Dutch Belted rabbit after <italic>in vivo</italic> excitation mapping. The imaged volume represents the retinal tissue with a diameter of 0.25 mm, spanning the inner nuclear, inner plexiform, and ganglion cell layers. Serial EM sections were cut at 70&#x02013;90 nm with a Leica UC6 ultramicrotome and captured at the resolution of 2.18 nm/pixel across both axes using SerialEM (Mastronarde, <xref ref-type="bibr" rid="B35">2005</xref>). In RC1, there are in total 341 EM mosaics generated by the NCR Toolset (Anderson et al., <xref ref-type="bibr" rid="B4">2009</xref>), and we clipped out 120 images in the size of 9,958 &#x000D7; 9,959 pixels from the randomly chosen sections.</p>
<p>To annotate cell membrane with high quality on the 120 images, we launched an iterative annotation project that lasted for 3 months. All the annotators were trained to recognize and annotate cellular membrane in EM images, but only two-thirds of all, 53 annotators, were finally qualified to participate in the project according to their annotation results. In the iterative annotation procedure, each EM image had undergone three continuous rounds of annotation with the guidance of blind review. The final round of annotation was regarded as the &#x0201C;ground truth.&#x0201D; While the first two rounds are valuable for analyzing the human learning process, we also reserved the intermediate results for public release. All of the U-RISC datasets are released at <ext-link ext-link-type="uri" xlink:href="https://github.com/EmmaSRH/U-RISC-Data-Code">https://github.com/EmmaSRH/U-RISC-Data-Code</ext-link>.</p>
</sec>
<sec>
<title>Competition</title>
<p>The goal of the competition was to predict cell membranes in the EM images of U-RISC. Participants were required to return images depicting the boundary of all neurons. F1-score was selected as the evaluation criterion for the accuracy of the results (Formula 1) (Sasaki and Fellow, <xref ref-type="bibr" rid="B41">2007</xref>). During the evaluation processing, according to the classes of prediction and ground truth, the predicted pixels of images were first divided into four types: true positive (TP), true negative (TN), false positive (FP), and false negative (FN). Then, two metrics, precision and recall, were calculated from the number of these types of pixels. The F1-score was defined as the harmonic mean of precision and recall.</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>F</mml:mtext><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mtext>score&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mtext>Precision</mml:mtext><mml:mo>&#x000D7;</mml:mo><mml:mtext>Recall</mml:mtext></mml:mrow><mml:mrow><mml:mtext>Precision</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>Recall</mml:mtext></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>Precision&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>FP</mml:mtext></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mtext>&#x000A0;Recall&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>FN</mml:mtext></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>There were two tracks in the competition; images in Track 1 were kept in their original size (9,958 &#x000D7; 9,959 pixels), images in Track 2 were downsampled to the size of 1,024 &#x000D7; 1,024 pixels. Fifty images, 30 as the training dataset and 20 as the test dataset, were released in Track 1. Additionally, Track 2 contained 70 images in total, amounting to 40 training images and 30 testing images. The training dataset included EM images with their corresponding ground truth, while the ground truth of the test dataset was kept private. In both the tracks, ten images from the training dataset served as the validation dataset for the participants to monitor and develop their models. No statistical methods were used to determine the assignment of images in the whole arrangement.</p>
</sec>
<sec>
<title>Segmentation Networks</title>
<p>We conducted experiments to compare the performance of the same methods on U-RISC (Track 2) and ISBI 2012. Three representative deep learning networks such as (<bold>Table 2</bold>), U-Net (Ronneberger et al., <xref ref-type="bibr" rid="B37">2015</xref>), LinkNet (Chaurasia and Culurciello, <xref ref-type="bibr" rid="B10">2017</xref>), and CASENet (Yu et al., <xref ref-type="bibr" rid="B59">2017</xref>) were considered. The three networks are all pixel-based segmentation networks. Specifically, given the input image <italic>x</italic>, the goal of the networks is to classify the corresponding semantic cell membrane pixel by pixel. For the input image <italic>x</italic> and the classification function <italic>F</italic>(<italic>x</italic>), <italic>Y</italic>{<italic>p</italic>|<italic>X</italic>, &#x00398;}, &#x00398; &#x02208; [0, 1] is taken as the output of the network, which represents the edge probability of the semantic category of the pixel <italic>p</italic>. &#x00398; are the parameters in the network and are optimized in the training process. Architectures of the three networks are described as follows.</p>
<sec>
<title>U-Net</title>
<p>The U-Net (Ronneberger et al., <xref ref-type="bibr" rid="B37">2015</xref>) is a classical fully convolutional network (i.e., there is no fully connected operation in the network). The model is composed of two parts: contracting path and expansive path. The contracting path follows the typical architecture of a convolutional network. At each downsampling step, the U-Net doubles the number of feature channels to gain a concatenation with the correspondingly cropped feature map from the contracting path. At the final layer, a 1 &#x000D7; 1 convolution is used to map each 64-component feature vector to the desired number of classes. In total, the network has 23 convolutional layers. We use ResNet50 as its encoder.</p>
</sec>
<sec>
<title>LinkNet</title>
<p>The model structure of LinkNet (Chaurasia and Culurciello, <xref ref-type="bibr" rid="B10">2017</xref>) is almost similar to the U-Net, which is a typical encoder&#x02013;decoder structure. The encoder starts with an initial block which performs convolution on the input image with a kernel of size 7 &#x000D7; 7 and a stride of 2. This block also performs spatial max-pooling in an area of 3 &#x000D7; 3 with a stride of 2. The later portion of the encoder consists of residual blocks and is represented as the encoder-block. To reduce parameters, the LinkNet uses ResNet18 as its encoder.</p>
</sec>
<sec>
<title>CASENet</title>
<p><bold>The</bold> CASENet (Yu et al., <xref ref-type="bibr" rid="B59">2017</xref>) is an end-to-end deep semantic edge learning architecture adopting ResNet-152 as its backbone. The classification module here consists of a 1 &#x000D7; 1 convolution and a bilinear interpolation upsampling layer to generate M active images; each image size is the same as the original image. Each residual block is followed by a classification module to obtain five classification activation graphs. Then, a sliced concatenation layer is used to fuse the M classification activation graphs, and finally, a 5M-channel activation graph is obtained. The activation graphs are used as the input for the fused classification layer to obtain an M-channel activation graph. The fusion classification layer is the convolution of the M group, 1 &#x000D7; 1.</p>
</sec>
<sec>
<title>Transfer Learning</title>
<p>The pretrained model from Conrad and Narayan (<xref ref-type="bibr" rid="B14">2021</xref>) was used in our method, specifically, MoCoV2 (Arar et al., <xref ref-type="bibr" rid="B6">2020</xref>) and CEM500K (Conrad and Narayan, <xref ref-type="bibr" rid="B14">2021</xref>) were respectively selected as the pretraining method and dataset.</p>
</sec>
<sec>
<title>Training Settings</title>
<p>For each dataset, the same training and testing data distribution was utilized for the three methods. For U-RISC, during the training, the original images were cut into 1,024 &#x000D7; 1,024 patches with overlaps. Additionally, the patches were randomly assigned to the training set and validation set according to the ratio of 50,000/20,000. For ISBI 2012, 20 images were used for training, and 10 images were used for testing.</p>
</sec>
<sec>
<title>Loss Function and Optimization</title>
<p>The U-RISC image membrane segmentation task can be defined as the pixel-level classification task. The ground truth of each pixel is a binary value <italic>y</italic> &#x02208; {0, 1}, and <italic>y</italic>&#x02032; is the predicted value by the prediction model. <italic>Y</italic> is the set of all pixels of one image. For each algorithm, we used the same loss function and optimization method. Specifically, focal loss and dice loss were chosen. Focal loss and dice loss are defined as:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mtext>Focal</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>Y</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow></mml:msup><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E4"><label>(4)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mtext>Dice</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>Y</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mi>y</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The final loss function is the summation of the two losses with the proportion of 1: &#x003BB;. That is <italic>L</italic> &#x0003D; <italic>L</italic><sub><italic>Focal</italic></sub> &#x0002B; &#x003BB;<italic>L</italic><sub><italic>Dice</italic></sub>. We set &#x003BB; &#x0003D; 1 and &#x003B3; &#x0003D; 2 in our experiments. When optimizing the parameters in the network, we chose Adam (Kingma and Ba, <xref ref-type="bibr" rid="B26">2014</xref>) as the optimizer.</p>
</sec>
<sec>
<title>Implementation Details</title>
<p>Data augmentation (random horizontal/vertical flip, random rotation, random zoom, random cropping, random cropping, random translation, random contrast, and random color jitter) was used. Four Nvidia V100 GPUs were used for training. In the testing stage, the original images were cut into the same size as the training images, and the patches were tested. These patches were eventually mosaiced back to the original size for evaluation. The parameter settings are shown in <xref ref-type="table" rid="T1">Table 1</xref>. Mean value and standard error are computed by testing the images of each dataset. The methods with &#x0201C;-<sup>&#x0002A;</sup>&#x0201D; in the table represent that they are implemented by us.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Implementation details.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Implementation</bold></th>
<th valign="top" align="center"><bold>U-Net-<sup>&#x0002A;</sup></bold></th>
<th valign="top" align="center"><bold>CASENet-<sup>&#x0002A;</sup></bold></th>
<th valign="top" align="center"><bold>LinkNet-<sup>&#x0002A;</sup></bold></th>
<th valign="top" align="center"><bold>U-Net-transfer</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Data augmentation</td>
<td valign="top" align="center">&#x0221A;</td>
<td valign="top" align="center">&#x0221A;</td>
<td valign="top" align="center">&#x0221A;</td>
<td valign="top" align="center">&#x0221A;</td>
</tr>
<tr>
<td valign="top" align="left">Pre-training</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">&#x0221A;</td>
</tr>
<tr>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="center">1e-3</td>
<td valign="top" align="center">1e-7</td>
<td valign="top" align="center">5e-4</td>
<td valign="top" align="center">2e-5</td>
</tr>
<tr>
<td valign="top" align="left">Batch size</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">4</td>
</tr>
<tr>
<td valign="top" align="left">GPUs</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">8</td>
</tr>
<tr>
<td valign="top" align="left">Epoch</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">300</td>
<td valign="top" align="center">50</td>
</tr>
<tr>
<td valign="top" align="left">Worker</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">32</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec>
<title>Image Definition Criteria</title>
<p>As the competition includes two tracks and the participants have obvious different performances on them, we introduced the four representative image definition criteria, Brenner (Subbarao and Tyan, <xref ref-type="bibr" rid="B46">1998</xref>), SMD2 (Thakkinstian et al., <xref ref-type="bibr" rid="B50">2005</xref>), Variance (Saltelli et al., <xref ref-type="bibr" rid="B40">2010</xref>), and Vollath (<xref ref-type="bibr" rid="B51">2008</xref>) to analyze the effects of downsampling on EM images (in discussion and <xref ref-type="supplementary-material" rid="SM7">Appendix Figure</xref>). The former two consider the difference and variance of gray values between adjacent pixels, while the latter two consider the whole image.</p>
<p>Brenner gradient function simply calculates the square of the gray difference between two adjacent pixels.</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mo>|</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where, <italic>f</italic>(<italic>x, y</italic>) represents the gray value of pixel (<italic>x, y</italic>) corresponding to image <italic>f</italic>, and <italic>D</italic>(<italic>f</italic>) is the result of image definition calculation (the same below).</p>
<p>The SMD2 multiplies two gray variances in each pixel field and then accumulates them one by one.</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mo>|</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The variance function is defined as</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msup><mml:mrow><mml:mo>|</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>-</mml:mo><mml:mi>&#x003BC;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BC; is the average gray value of the whole image, which is sensitive to noise. The purer the image, the smaller is the function value.</p>
<p>The Vollrath function is defined as follows:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>M</mml:mi><mml:mi>N</mml:mi><mml:msup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BC; is the average gray value of the whole image, <italic>M</italic> and <italic>N</italic> are the width and height of the image, respectively.</p>
</sec>
<sec>
<title>Attribution Analysis</title>
<p>We also noticed the different performance of U-Net when applied on ISBI 2012 and U-RISC. To explore the deeper reason, we carry out an attribution analysis on the U-Net by using the IG (Sundararajan et al., <xref ref-type="bibr" rid="B47">2017</xref>) method to quantify the contribution maps (in section Attribution Analysis of the Deep Learning Method on U-RISC and ISBI 2012). For a given input image <italic>x</italic> and model <italic>F</italic>(<italic>x</italic>), the goal of the network is to find out which pixels or features in <italic>x</italic> have an important influence on the decision-making of the model or sort the importance of each pixel or feature in <italic>x</italic>. Such a process is defined as attribution. The IG uses the integrated value along the whole gradient line from the input to the output. In the cell membrane segmentation task, from the decision of a pixel of <italic>y</italic> (predicted as the cell membrane or not), we can obtain the contribution of each pixel of the input image. Putting the contribution of each pixel together, we record it as an attribution field <italic>A</italic>, whose size is the same as the original image. The value <italic>x</italic><sub><italic>i</italic></sub> denotes the <italic>i</italic><sub><italic>th</italic></sub> pixel in image <italic>x</italic>, and <italic>w</italic><sub><italic>i</italic></sub> denotes the attribution value of <italic>x</italic><sub><italic>i</italic></sub>, representing the contribution decision of pixel <italic>x</italic><sub><italic>i</italic></sub> to <italic>y</italic>. The value of <italic>w</italic><sub><italic>i</italic></sub> is normalized to [&#x02212;1,1].</p>
<p>In the binary segmentation task, for the current input image <italic>x</italic>, if we know that the output <italic>y</italic> is a specific value, such as <italic>y</italic> &#x0003D; 0, and the corresponding reference image is <italic>x</italic>&#x02032;, then we can take a linear interpolation, i.e.,</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>If the constant &#x003B1; &#x0003D; 0, then the input image is the base image as that of <italic>x</italic>&#x02032;. If &#x003B1; &#x0003D; 1, then the input image is the current image, which is <italic>x</italic>. When 0 &#x0003C; &#x003B1; &#x0003C; 1, it can be other images.</p>
<p>For the output of the neural network <italic>F</italic>(<italic>x</italic>), the attribution value of <italic>x</italic><sub><italic>i</italic></sub>, <italic>w</italic><sub><italic>i</italic></sub> is computed as follows.</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>F</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mi>d</mml:mi><mml:mi>&#x003B1;</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, <inline-formula><mml:math id="M16"><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>F</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mtext>&#x000A0;</mml:mtext></mml:math></inline-formula>is the gradient of <italic>F</italic>(<italic>x</italic>) with respect to <italic>x</italic><sub><italic>i</italic></sub>.</p>
<p>As the resolution and image size of U-RISC and ISBI 2012 are different, for a fair comparison, we define the size of the pixel attribution field as <italic>S</italic><sub><italic>k</italic></sub>, which represents the physical size corresponding to the pixel area with the fixed contribution value threshold, <italic>k</italic>. If the attribution value <italic>w</italic><sub><italic>i</italic></sub> is greater than <italic>k</italic>, the pixel is the one with a higher contribution in decision-making. The area of the attribution field <italic>S</italic><sub><italic>k</italic></sub> is obtained by multiplying the number of pixels with the attribution value, <italic>w</italic><sub><italic>i</italic></sub> which is larger than <italic>k</italic> and the corresponding physical size of the pixel (square of resolution <italic>h</italic>).</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0003E;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>A</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>Data Analysis</title>
<p>All statistical tests used, including statistic values and sample sizes, are provided in the figure captions, including the mean and standard. All analyses were performed using custom software developed using the following tools and software: MATLAB (R2018a), Python (3.6), PyTorch (1.6.0), NumPy (1.19.0), SciPy (1.5.1), and matplotlib (2.2.3).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>The Largest Ultra-High-Resolution EM Cell Membrane Segmentation Dataset</title>
<p>Along with this article, we proposed a new EM dataset with cell membrane annotated the U-RISC. To our best knowledge, U-RISC has the highest resolution among the publicly available annotated EM datasets (refer to <xref ref-type="fig" rid="F2">Figure 2A</xref> as an example). It was annotated upon the rabbit retinal connectomic dataset RC1 (Anderson et al., <xref ref-type="bibr" rid="B5">2011</xref>) with a 2.18 nm/pixel resolution at both the <italic>x</italic> and <italic>y</italic> axes. Taking ISBI 2012 as an example (<xref ref-type="fig" rid="F2">Figure 2B</xref>) (120 pairs of 9,958 &#x000D7; 9,959 pixel images in the U-RISC and 30 pairs of 512 &#x000D7; 512 pixel images in ISBI 2012) (<xref ref-type="fig" rid="F2">Figure 2C</xref>). One characteristic of U-RISC is that cell membranes only cover a small area of the images, making it an imbalanced dataset for deep learning (an average of 5.10% &#x000B1; 2% in U-RISC compared to 21.65% &#x000B1; 2% in ISBI 2012).</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Comparison between U-RISC and ISBI 2012. <bold>(A,B)</bold> An example of U-RISC and ISBI 2012 data includes the raw EM image (top) and the corresponding annotation result (bottom). Black pixels in annotation results represent cellular membranes. <bold>(C)</bold> (Top) Both the number and size of images in U-RISC surpass those in ISBI 2012. (Bottom) The proportion of annotated pixels, 21.65 &#x000B1; 2% in ISBI 2012 and 5.10 &#x000B1; 2% in U-RISC, making the latter a more imbalanced dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0002.tif"/>
</fig>
<p>We employed an iterative manual annotation procedure to ensure the quality of annotation. Because of the difficulty in distinguishing the cell membrane from the organelle membrane, special attention was paid to exclude the organelle membrane from annotation (<xref ref-type="fig" rid="F3">Figure 3A</xref>). In practical connectomic research, the image quality can be affected by many reasons, such as insufficient staining and thick section. Considering this, we retained several images with low quality in the U-RISC to make the dataset closer to the actual situation. Annotation on these images costs more time and caution (<xref ref-type="fig" rid="F3">Figure 3B</xref>). Labeling errors could be detected and then corrected in each round of iteration (<xref ref-type="fig" rid="F4">Figure 4</xref>). For scientific research reasons, the human labeling process is very valuable for uncovering the human learning process. Therefore, the intermediate annotated results were also reserved for public release (<ext-link ext-link-type="uri" xlink:href="https://github.com/EmmaSRH/U-RISC-Data-Code">https://github.com/EmmaSRH/U-RISC-Data-Code</ext-link>).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Examples of images with their annotations. <bold>(A)</bold> Organelle membranes were cautiously avoided to be annotated. <bold>(B)</bold> More time and patience were needed to annotate the image with low contrast.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0003.tif"/>
</fig>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Example of iterative human annotation. <bold>(A)</bold> Original image to be annotated. <bold>(B)</bold> Many errors were found in the first round of annotation. <bold>(C)</bold> After correction, much fewer errors were detected in the second round of annotation, and the correction results were served as the final annotation. Red small triangles and boxes indicate false-positive errors (enlargement in the bottom left), blue for false-negative errors (enlargement in the bottom right).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0004.tif"/>
</fig>
</sec>
<sec>
<title>Ultra-High Resolution EM Images Segmentation Competition</title>
<p>To investigate the performance of the deep learning methods on the U-RISC and to propose a benchmark, a competition on cellular membrane segmentation was organized by the Beijing Academy of Artificial Intelligence Institution, Beijing, China (BAAI) and the Peking University, Beijing, China (PKU)<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref>. In total, 448 participants took part in the competition, mainly from domestic competitive universities, research organizations, and top IT institutions.</p>
<p>There were two tracks in the competition (<xref ref-type="table" rid="T2">Table 2</xref>): Track 1 used the original images with the size of 9,958 &#x000D7; 9,959 pixels as training and testing datasets, respectively. In Track 2, the images were downsampled to the size of 1,024 &#x000D7; 1,024 pixels. The purpose of Track 2 was to allow researchers with limited computational resources to participate in the competition. The final round of human annotation was used as the ground truth to evaluate the algorithms, and an F1-score was applied as the evaluation metric (for details, please refer to Methods and Materials).</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Leaderboard of track 1 and track 2.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="center" colspan="3" style="border-bottom: thin solid #000000;"><bold>Track 1 (original)</bold></th>
<th valign="top" align="center" colspan="3" style="border-bottom: thin solid #000000;"><bold>Track 2 (downsample)</bold></th>
</tr>
<tr>
<th valign="top" align="left"><bold>Team name</bold></th>
<th valign="top" align="left"><bold>Institution</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
<th valign="top" align="left"><bold>Team name</bold></th>
<th valign="top" align="left"><bold>Institution</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Human 1st</td>
<td valign="top" align="left">&#x02013;</td>
<td valign="top" align="center">0.92128 &#x000B1; 0.012</td>
<td valign="top" align="left">Human 1st</td>
<td valign="top" align="left">&#x02013;</td>
<td valign="top" align="center">0.96915 &#x000B1; 0.014</td>
</tr>
<tr>
<td valign="top" align="left">Human 2nd</td>
<td valign="top" align="left">&#x02013;</td>
<td valign="top" align="center">0.92128 &#x000B1; 0.012</td>
<td valign="top" align="left">Human 2nd</td>
<td valign="top" align="left">&#x02013;</td>
<td valign="top" align="center">0.99891 &#x000B1; 0.003</td>
</tr>
<tr>
<td valign="top" align="left">SCP173</td>
<td valign="top" align="left">Tencent<xref ref-type="table-fn" rid="TN1"><sup>a</sup></xref></td>
<td valign="top" align="center">0.60704 &#x000B1; 0.043</td>
<td valign="top" align="left">Horch</td>
<td valign="top" align="left">UCAS<xref ref-type="table-fn" rid="TN2"><sup>b</sup></xref></td>
<td valign="top" align="center">0.56932 &#x000B1; 0.053</td>
</tr>
<tr>
<td valign="top" align="left">yangsenwxy</td>
<td valign="top" align="left">SCU<xref ref-type="table-fn" rid="TN3"><sup>c</sup></xref></td>
<td valign="top" align="center">0.60701 &#x000B1; 0.042</td>
<td valign="top" align="left">Deadline</td>
<td valign="top" align="left">NJU<xref ref-type="table-fn" rid="TN4"><sup>d</sup></xref></td>
<td valign="top" align="center">0.56213 &#x000B1; 0.055</td>
</tr>
<tr>
<td valign="top" align="left">SpongeBobbb</td>
<td valign="top" align="left">HDU<xref ref-type="table-fn" rid="TN5"><sup>e</sup></xref></td>
<td valign="top" align="center">0.60480 &#x000B1; 0.042</td>
<td valign="top" align="left">SpongeBobbb</td>
<td valign="top" align="left">HDU<xref ref-type="table-fn" rid="TN5"><sup>e</sup></xref></td>
<td valign="top" align="center">0.56136 &#x000B1; 0.049</td>
</tr>
<tr>
<td valign="top" align="left">VIDAR</td>
<td valign="top" align="left">USTC<xref ref-type="table-fn" rid="TN6"><sup>f</sup></xref></td>
<td valign="top" align="center">0.60303 &#x000B1; 0.041</td>
<td valign="top" align="left">VIDAR</td>
<td valign="top" align="left">USTC<xref ref-type="table-fn" rid="TN6"><sup>f</sup></xref></td>
<td valign="top" align="center">0.55170 &#x000B1; 0.046</td>
</tr>
<tr>
<td valign="top" align="left">Deadline</td>
<td valign="top" align="left">NJU<sup>d5</sup></td>
<td valign="top" align="center">0.60066 &#x000B1; 0.045</td>
<td valign="top" align="left">Archer</td>
<td valign="top" align="left">THU<xref ref-type="table-fn" rid="TN7"><sup>g</sup></xref></td>
<td valign="top" align="center">0.55107 &#x000B1; 0.047</td>
</tr>
<tr>
<td valign="top" align="left">Chasingstar</td>
<td valign="top" align="left">JLU<xref ref-type="table-fn" rid="TN8"><sup>h</sup></xref></td>
<td valign="top" align="center">0.59647 &#x000B1; 0.044</td>
<td valign="top" align="left">scu_ws</td>
<td valign="top" align="left">SCU<xref ref-type="table-fn" rid="TN3"><sup>c</sup></xref></td>
<td valign="top" align="center">0.54847 &#x000B1; 0.053</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN1">
<label>a</label>
<p><italic>Tencent Holdings Ltd (China)</italic>.</p></fn>
<fn id="TN2">
<label>b</label>
<p><italic>University of Chinese Academy of Sciences (China)</italic>.</p></fn>
<fn id="TN3">
<label>c</label>
<p><italic>Sichuan University (China)</italic>.</p></fn>
<fn id="TN4">
<label>d</label>
<p><italic>Nanjing University (China)</italic>.</p></fn>
<fn id="TN5">
<label>e</label>
<p><italic>Hangzhou Dianzi University (China)</italic>.</p></fn>
<fn id="TN6">
<label>f</label>
<p><italic>University of Science and Technology of China</italic>.</p></fn>
<fn id="TN7">
<label>g</label>
<p><italic>Tsinghua University (China)</italic>.</p></fn>
<fn id="TN8">
<label>h</label>
<p><italic>Jilin University (China)</italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>Surprisingly, from the competition, the top 6 teams in each track gained F1-scores around 0.6 on U-RISC, which were far below the human levels (0.92 and 0.99, the first and second rounds of annotation). However, a previous study has shown that the performance of the top teams in ISBI 2012 had already been reasonably closer to the human level (Arganda-Carreras et al., <xref ref-type="bibr" rid="B7">2015</xref>). To investigate the causes of the performance gap between the methods and humans on the U-RISC, we first surveyed the top 6 teams in our competition. It indicated that a variety of current popular approaches to segmentation were utilized (<xref ref-type="fig" rid="F5">Figure 5</xref>). From the choice of models (<xref ref-type="fig" rid="F5">Figure 5A</xref>), the participants used the current popular image segmentation networks, such as U-Net (Ronneberger et al., <xref ref-type="bibr" rid="B37">2015</xref>), Efficientnet (Tan and Le, <xref ref-type="bibr" rid="B49">2019</xref>), and CASENet (Yu et al., <xref ref-type="bibr" rid="B59">2017</xref>). For backbone selection, the ResNet (He et al., <xref ref-type="bibr" rid="B23">2016</xref>) and their variants were the most chosen architectures. Data augmentation was ubiquitously applied to improve the generalization of the models. About 13% of the participants used Hypercolumns (Hariharan et al., <xref ref-type="bibr" rid="B22">2015</xref>) to improve the expressiveness of the model. From the design of the loss function, functions that can adjust penalty ratios according to sample distributions were applied to reduce the effect of sample imbalance, such as dice loss (Dice, <xref ref-type="bibr" rid="B17">1945</xref>), focal loss (Lin et al., <xref ref-type="bibr" rid="B31">2017</xref>), and BCE loss (Cui et al., <xref ref-type="bibr" rid="B15">2019</xref>). Additionally, Adam (Kingma and Ba, <xref ref-type="bibr" rid="B26">2014</xref>) was shown to be the most chosen optimization method.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Mean F1-scores of teams with different methods used. <bold>(A,B)</bold> The statistics of Track 1 and Track 2, respectively. The x-axis represents the proportion of the team with the method, the y-axis represents the average of F1-scores.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0005.tif"/>
</fig>
<p>The analysis suggested that even though participants had considered many popular methods, their performance was still not satisfactory and varied only slightly between each other. To identify whether this was because of the challenges of U-RISC or the methods themselves, we picked out the three widely used methods, the U-Net (Ronneberger et al., <xref ref-type="bibr" rid="B37">2015</xref>), LinkNet (Chaurasia and Culurciello, <xref ref-type="bibr" rid="B10">2017</xref>), and CASENet (Yu et al., <xref ref-type="bibr" rid="B59">2017</xref>). We conducted a fair comparison between the performance of each method on U-RISC (Track 1) and ISBI 2012. Results showed that these methods could reach over 0.97 (F1-score) in ISBI 2012, but only between 0.57 and 0.61 in the U-RISC (<xref ref-type="table" rid="T3">Table 3</xref>), which confirmed that the performance gap in competition comes from the challenges of U-RISC.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>F1-scores in U-RISC and ISBI 2012.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>U-RISC</bold></th>
<th valign="top" align="center"><bold>ISBI 2012</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">LinkNet-<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.60701 &#x000B1; 0.063</td>
<td valign="top" align="center">0.97246 &#x000B1; 0.08</td>
</tr>
<tr>
<td valign="top" align="left">CASENet-<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.60065 &#x000B1; 0.053</td>
<td valign="top" align="center">0.97132 &#x000B1; 0.08</td>
</tr>
<tr>
<td valign="top" align="left">U-Net-<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.57123 &#x000B1; 0.049</td>
<td valign="top" align="center">0.97010 &#x000B1; 0.09</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>What are the unique challenges brought by U-RISC to deep learning algorithms? Two types of errors were analyzed first: false-positive errors, which led to incorrect membrane predictions, and false-negative errors, which caused incontinuity in the cell membrane. According to our analysis, both false-positive errors (pink boxes) and false-negative errors (orange boxes) were common in the U-RISC, which were rare in ISBI 2012 (<xref ref-type="fig" rid="F6">Figures 6B,C</xref>). More examples can be found in <xref ref-type="fig" rid="F6">Figure 6</xref> and <xref ref-type="supplementary-material" rid="SM1">Supplementary Figures 1</xref>&#x02013;<xref ref-type="supplementary-material" rid="SM3">3</xref>. Further investigations for the networks are required to explore the reason and find ways to reduce the errors.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Errors in segmentation predictions of U-RISC and ISBI 2012. <bold>(A)</bold> The examples of false-positive and false-negative errors. <bold>(B,C)</bold> The examples of two errors in the segmentations of U-RISC and ISBI 2012. Pink arrows and lines represent false-negative errors, and orange represents false-positive errors.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0006.tif"/>
</fig>
</sec>
<sec>
<title>Attribution Analysis of the Deep Learning Method on U-RISC and ISBI 2012</title>
<p>To acquire a deeper understanding of the different performances in U-RISC and ISBI 2012, we performed an attribution analysis (Ancona et al., <xref ref-type="bibr" rid="B3">2019</xref>) on the trained U-Net. We selected the gradient-based attribution method, the IG (Sundararajan et al., <xref ref-type="bibr" rid="B47">2017</xref>), which is widely applied to explainable artificial intelligence, such as understanding feature importance (Adadi and Berrada, <xref ref-type="bibr" rid="B1">2018</xref>), identifying data skew (Clark et al., <xref ref-type="bibr" rid="B13">2019</xref>), and debugging model performance (Guidotti et al., <xref ref-type="bibr" rid="B21">2018</xref>). In brief, IG aims to explain the relationship between predictions and input features based on gradients (<xref ref-type="fig" rid="F7">Figure 7A</xref>). The IG output is plotted in Attribution Fields to reflect their contribution to the final prediction. In the heatmap, each pixel was assigned with a normalized value between [&#x02212;1, 1]. With IG, we analyzed the attribution field of each predicted pixel of U-Net in U-RISC and ISBI 2012. Color and shade were used to represent the normalized contribution values in attribution fields (<xref ref-type="fig" rid="F7">Figure 7B</xref>). For a fair comparison between U-RISC and ISBI 2012, areas of pixel attribution fields, <italic>S</italic><sub><italic>k</italic></sub> were converted to physical size according to their respective resolutions.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Attribution analysis. <bold>(A)</bold> Integrated gradients (IG) attribution method. <bold>(B)</bold> Statistics of attribution filed for U-RISC and ISBI 2012.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0007.tif"/>
</fig>
<p><xref ref-type="fig" rid="F8">Figure 8</xref> shows the examples of attribution fields, where bounding boxes with different colors represented different pixel classifications, green for a correct predicted pixel, orange for a false-positive error, and pink for a false-negative error. More examples can be found in <xref ref-type="supplementary-material" rid="SM4">Supplementary Figures 4</xref>&#x02013;<xref ref-type="supplementary-material" rid="SM6">6</xref>. We noticed that the areas of attribution fields <italic>S</italic><sub><italic>k</italic></sub> of two datasets were both relatively minor to the whole images (<xref ref-type="fig" rid="F7">Figure 7B</xref>). For example, at the threshold of <italic>k</italic> &#x0003E; 0.01, the <italic>S</italic><sub><italic>k</italic></sub> of the correct cases accounted for only 5.1 and 0.8% relative to the whole image (the green bounding boxes in <xref ref-type="fig" rid="F8">Figure 8</xref>). This suggested that the U-Net would focus on local characteristics within small areas of the images when making predictions. In addition, we found that the averaged <italic>S</italic><sub><italic>k</italic></sub> of each predicted pixel in U-RISC was significantly larger than that in ISBI 2012, specifically 46,000 <italic>n</italic>m<sup>2</sup> in U-RISC and 10,300 <italic>n</italic>m<sup>2</sup> in ISBI 2012. Taken together, the U-Net would predict cell membrane according to local information around the pixel, and the average attribution field was larger in U-RISC than that in ISBI 2012. All of these indicate that more information is required for the segmentation in U-RISC.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Attribution analysis. <bold>(A,B)</bold> Attribution fields of ISBI 2012 and U-RISC dataset. The first line represents the original image, network prediction result, and annotation respectively. The pixels pointed by green (correct cell membrane pixel), orange (false-positive predicted pixel), and pink (false-negative predicted pixel) arrows are the prediction points used in the attribution method. Images in the three-color boxes with the same size in the second line represent the attribution field corresponding to the above three pixels. Blue indicates that the network is likely to predict the pixels as the cell membrane, while the opposite is indicated by red.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0008.tif"/>
</fig>
</sec>
<sec>
<title>U-Net-Transfer Model Achieves the SOTA Result on the U-RISC Benchmark</title>
<p>Considering both the comprehensive analyses of competition and attribution analysis, we integrated outstanding methods to develop our method (<xref ref-type="fig" rid="F9">Figure 9A</xref>). For basic segmentation architecture, we chose the U-Net due to its better characteristic extraction ability. Many valuable techniques were also considered, including a cross-crop strategy for saving computational resources and data augmentation to increase data diversity. We chose both focal loss and dice loss to deal with the imbalance of samples for the loss function design. Some parameters used for training were also optimized, such as batch-size/GPU (4) and the number of GPUs (8). For more details, please refer to Segmentation networks in Methods and materials. Especially, a recent study has shown that transfer learning with domain-specific annotated datasets could be effective in elevating deep learning models&#x00027; performance (Conrad and Narayan, <xref ref-type="bibr" rid="B14">2021</xref>). Therefore, we introduced a pretrained model, trained with MoCoV2 (Arar et al., <xref ref-type="bibr" rid="B6">2020</xref>) on CEM500K (Conrad and Narayan, <xref ref-type="bibr" rid="B14">2021</xref>). The segmentation result showed that the F1-scores of our method were 10% higher than the leader of the competition (0.66 vs. 0.61 in <xref ref-type="table" rid="T2">Table 2</xref> and <xref ref-type="fig" rid="F9">Figure 9B</xref>). Thus, we provide a new benchmark on the cellular membrane segmentation of U-RISC.</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>The U-Net-transfer method achieves the best performance on U-RISC. <bold>(A)</bold> The pretraining, training, and testing processing for U-Net. <bold>(B)</bold> The comparison of the F1-scores. &#x0201C;SCP-173&#x0201D; represents the top performance in the competition. U-Net-<sup>&#x0002A;</sup> represents the performance in <xref ref-type="table" rid="T2">Table 2</xref>. This section represents the performance of the U-Net-transfer method.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-16-842760-g0009.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>This article first proposed the U-RISC, a cell membrane EM dataset created through intensive and elaborate annotation. The dataset is characterized by the highest resolution and the largest single image size compared to the other current publicly available annotated EM datasets. Next, we organized a segmentation competition on U-RISC and proposed the benchmark. During the competition, we noticed that the performances of popular deep learning methods were far below that of humans, which motivated us to explore the causes. Thus, we carried out a comprehensive survey of the participants in the deep learning methods applied in the competition. To our surprise, methods, such as U-Net, LinkNet, and CASENet exhibited a significant drop of F1-score on the U-RISC compared to ISBI 2012, from 0.9 to 0.6. To explore the mechanisms underlying this discrepant performance, we introduced a gradient-based attribution method, the IG. Through attribution analysis of U-Net, we found that the average pixel attribution field of U-RISC is larger than that of ISBI, corresponding to the size of cellular structure, and both of them are relatively small to the whole image size. By integrating currently available methods, we improve the benchmark to 0.67, about 10% higher than the top leader from the competition. Based on the analyses in this article, here, we raise some considerations in the challenges for deep learning-based segmentation algorithms brought by U-RISC and propose several suggestions for improving the EM segmentation methods.</p>
<sec>
<title>Challenges for Deep Learning-Based Segmentation</title>
<p>Benchmark showed that the segmentation performance of deep learning algorithms on U-RISC was still far behind the human level. The U-RISC poses challenges for deep learning-based segmentation in the following aspects: (1) high computational costs needed to deal with large images, (2) the extreme sample imbalance caused by the low ratio of cellular membrane pixels in the whole image, and (3) side effects of typical data processing methods.</p>
<p>Deep learning itself is already a computationally intensive method. It would require more computational resources to process the images with a much larger size in the U-RISC. In practical terms, taking U-Net as an example, processing a 1,024 &#x000D7; 1,024 pixel image requires a GPU with 12GB memory. This memory is enough to deal with the images in ISBI 2012, of which the size is 512 &#x000D7; 512 pixels. But the size of a single image in the U-RISC is 9,958 &#x000D7; 9,959 pixels, which is far beyond the processing ability of the commonly used 12 GB memory GPU. Therefore, the additional computational burden brought by the U-RISC raises the first challenge for deep learning-based segmentation.</p>
<p>The problem of imbalanced samples widely exists in computational vision tasks (Li et al., <xref ref-type="bibr" rid="B30">2010</xref>; Alejo et al., <xref ref-type="bibr" rid="B2">2016</xref>; Zhang et al., <xref ref-type="bibr" rid="B60">2020</xref>), which should be considered when designing algorithms. Cellular membrane segmentation is a typical situation of sample imbalance because the cellular membrane only occupies a small proportion of the whole cell structure. According to statistics, the pixels belonging to the cellular membrane account for 21.65% of the entire pixels of ISBI 2012. While the proportion in U-RISC is much smaller, 5.10%, making the U-RISC an extremely imbalanced dataset. Preexisting solutions were mainly proposed from several aspects: loss function design (Lin et al., <xref ref-type="bibr" rid="B31">2017</xref>; Cui et al., <xref ref-type="bibr" rid="B15">2019</xref>), data augmentation (Yoo et al., <xref ref-type="bibr" rid="B58">2020</xref>), under/over-sampling (Fern&#x000E1;ndez et al., <xref ref-type="bibr" rid="B18">2018</xref>), and semantically multi-modal approaches (Zhu et al., <xref ref-type="bibr" rid="B61">2020</xref>). However, even though the participants in the competition already used these approaches, the final results showed a limited improvement in segmentation. So, the imbalanced problem of U-RISC is yet to be solved and becomes another challenge for deep learning-based segmentation.</p>
<p>Proper data processing is essential and helpful for deep-learning algorithms. For example, a downsampling process on raw images with an enormous size is commonly adopted in the segmentation tasks (Thakkinstian et al., <xref ref-type="bibr" rid="B50">2005</xref>; Chen et al., <xref ref-type="bibr" rid="B11">2014</xref>). In Track 2 of our competition, we used the downsampled dataset to reduce the computational consumption, as usual. Surprisingly, we found that the F1-score of the same method dropped and the overall performance was also decreased in Track 2 compared to Track 1. We speculated that the key reason might be the degradation of image quality from Track 1 to Track 2. We confirmed the quality reduction through four representative indices, including Brenner (Subbarao and Tyan, <xref ref-type="bibr" rid="B46">1998</xref>), SMD2 (Thakkinstian et al., <xref ref-type="bibr" rid="B50">2005</xref>), Variance (Saltelli et al., <xref ref-type="bibr" rid="B40">2010</xref>), and Vollath (<xref ref-type="bibr" rid="B51">2008</xref>) (shown in <xref ref-type="supplementary-material" rid="SM7">Appendix Figure</xref>). More cautions should be paid when using traditional data processing methods, and more advanced data processing theories are expected from this point of view.</p>
</sec>
<sec>
<title>Suggestions for the Improvement of Segmentation Methods</title>
<p>To some degree, increasing computational resources are possible ways to cope with the challenges mentioned above. However, it might not be easy for all the community researchers to access sufficient computational power; therefore, innovations in algorithms are still crucial for our future success. To improve the performance of deep learning in EM segmentation, we provide several suggestions for developing deep-learning algorithms from the following perspectives: model design, training techniques, data processing, loss function design, and visualization tools.</p>
<sec>
<title>Model Design</title>
<p>As shown in the attribution analysis, the current models for segmentation, such as U-Net (Ronneberger et al., <xref ref-type="bibr" rid="B37">2015</xref>), Efficientnet (Tan and Le, <xref ref-type="bibr" rid="B49">2019</xref>), and CASENet (Yu et al., <xref ref-type="bibr" rid="B59">2017</xref>), are designed to focus on the local information to make predictions. However, in a high-resolution image, other structures, like organelle membrane and synaptic vesicles, might share similar features with the cellular membrane on a local scale, which leads to false-positive results. Additionally, this constitutes one of the major error types in the competition. Therefore, it might not be enough for the classifiers of a model to make correct decisions with only local features. Multi-scale features can increase the learning ability of the neural network, and studies have shown that models using global information could improve the performance greatly (Liu et al., <xref ref-type="bibr" rid="B34">2018</xref>, <xref ref-type="bibr" rid="B33">2020</xref>; Chen et al., <xref ref-type="bibr" rid="B12">2019</xref>). Therefore, more global information could also be considered in the future design of the segmentation network.</p>
</sec>
<sec>
<title>Training Techniques</title>
<p>Skillful training techniques can also be helpful in improving segmentation performance. According to our survey, a two-stage training strategy could be much better than a single-stage training strategy. A recent study also suggests that pretraining with domain-specific datasets can help network learning domain features (Conrad and Narayan, <xref ref-type="bibr" rid="B14">2021</xref>). Besides that, much experience can be learned from the existing training methods. The Hypercolumns module (Hariharan et al., <xref ref-type="bibr" rid="B22">2015</xref>) is used to accelerate the convergence of training by combining features at different scales, and the combination of features from different scales can help bring in global information. The ScSE (Roy et al., <xref ref-type="bibr" rid="B38">2018</xref>) module introduces an attention mechanism into the network, thus, bringing in global information. Hybrid architectures can also be considered because of their ability to expand the receptive field (Goceri, <xref ref-type="bibr" rid="B19">2019</xref>). In a word, improvement can be made at the phase of the training by utilizing advanced training techniques.</p>
</sec>
<sec>
<title>Data Processing</title>
<p>Data processing is commonly used in deep learning, while traditional downsampling methods were shown to have side effects in the competition. To alleviate the side effects, some quality enhancing methods for downsampled images could be expected, such as edge and region-based image interpolation algorithms (Hwang and Lee, <xref ref-type="bibr" rid="B25">2004</xref>; Asuni and Giachetti, <xref ref-type="bibr" rid="B8">2008</xref>), low bit rate-based approaches (Lin and Dong, <xref ref-type="bibr" rid="B32">2006</xref>; Wu et al., <xref ref-type="bibr" rid="B56">2019</xref>), and quality assessment research (Wang et al., <xref ref-type="bibr" rid="B54">2003</xref>; Wang and Bovik, <xref ref-type="bibr" rid="B53">2006</xref>; Vu et al., <xref ref-type="bibr" rid="B52">2018</xref>). Meanwhile, other data processing methods can also be taken into account. For example, in data augmentation, by augmenting the training data randomly (such as multi-scale and multi-angle), the dependence of the model on specific attributes can be reduced, which can be beneficial in EM segmentation with many imbalanced samples.</p>
</sec>
<sec>
<title>Loss Function Design</title>
<p>Loss function design is another important part of deep learning. But many current loss functions have their own disadvantages in our competition. For example, dice loss (Dice, <xref ref-type="bibr" rid="B17">1945</xref>) was designed to optimize F1-score directly, without consideration of data imbalance. Focal loss (Lin et al., <xref ref-type="bibr" rid="B31">2017</xref>) and BCE loss (Cui et al., <xref ref-type="bibr" rid="B15">2019</xref>) were used in the competition to care more about data imbalance by giving different penalties according to sample difficulty, but the improvement was limited as shown by the results. A better design of loss function should take an overall consideration of both the sample imbalance and evaluation criteria. Most of the common evaluation criteria, such as the F1-score, a pixel-based statistic, are inconsistent with the human subjective feeling to some extent. It might be a major cause of the performance gap between humans and algorithms. Some other structure-based criteria have appeared, such as V-Rand and V-info (Arganda-Carreras et al., <xref ref-type="bibr" rid="B7">2015</xref>) that integrate skeleton information of cell membrane and ASSD (Heimann et al., <xref ref-type="bibr" rid="B24">2009</xref>), considering the distance of point sets.</p>
</sec>
<sec>
<title>Visualization Tools</title>
<p>Visualization tools can help us have a better understanding of the network. In this article, from IG, we could learn the attribution fields of U-Net from the view of gradient, which inspires us to improve deep learning methods by paying more attention to global information. In comparison, many other visualization tools start from other characteristics of the network. Layer-wise relevance propagation (LRP) (Bach et al., <xref ref-type="bibr" rid="B9">2015</xref>) and deep Taylor decomposition (DTD) (Montavon et al., <xref ref-type="bibr" rid="B36">2017</xref>) get attribution distribution by modifying the propagation rules. The information-based method, the IBA (Schulz et al., <xref ref-type="bibr" rid="B42">2020</xref>) restricts the flow of information to accomplish attribution fields. Combining different visualization tools can help promote much more insightful inspiration in improving deep learning methods.</p>
<p>Overall, we provide an annotated EM cellular membrane dataset, U-RISC, and its benchmark. This indeed brings many challenges in deep learning and promotes the development of deep learning methods for segmentation.</p>
</sec>
</sec>
</sec>
<sec sec-type="data-availability" id="s5">
<title>Data Availability Statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://github.com/EmmaSRH/U-RISC-Data-Code">https://github.com/EmmaSRH/U-RISC-Data-Code</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>RS, WW, KD, and TJ contributed to conception and design of the study. RS, WW, LH, and ZL organized the database. All authors contributed to manuscript revision, read, and approved the submitted version.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>This work was partially supported by the Natural Science Foundation of China under contract 62088102.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s8">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack><p>We would like to thank Prof. Bryan William Jones for providing the rabbit retinal connectome. We also acknowledge High-Performance Computing Platform of Peking University for providing computational resources.</p>
</ack>
<sec sec-type="supplementary-material" id="s9">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fncom.2022.842760/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fncom.2022.842760/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Image_1.jpg" id="SM1" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 1</label>
<caption><p>Supplements for segmentation predictions of U-RISC.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_2.jpg" id="SM2" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 2</label>
<caption><p>Supplements for segmentation predictions of ISBI 2012.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_3.jpg" id="SM3" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 3</label>
<caption><p>Supplements for segmentation predictions of U-RISC (Track 2).</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_4.jpg" id="SM4" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 4</label>
<caption><p>Supplements for attribution analysis on ISBI 2012.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_5.jpg" id="SM5" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 5</label>
<caption><p>Supplements for attribution analysis on U-RISC.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_6.jpg" id="SM6" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 6</label>
<caption><p>Supplements for attribution analysis on U-RISC (Track 2).</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM7" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Appendix Figure</label>
<caption><p>Differences between the original image and downsampled image. <bold>(A)</bold> The crop of the original image. <bold>(B)</bold> The crop of downsampled images at the same position. The gray-scale histograms are calculated on A and B. <bold>(C)</bold> The scores of definition indices calculated on the whole U-RISC dataset before and after downsampling. Details of indices are described in Methods and materials.</p></caption>
</supplementary-material>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Adadi</surname> <given-names>A.</given-names></name> <name><surname>Berrada</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Peeking inside the black-box: a survey on explainable artificial intelligence (XAI)</article-title>. <source>IEEE Access</source> <volume>6</volume>, <fpage>52138</fpage>&#x02013;<lpage>52160</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2018.2870052</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alejo</surname> <given-names>R.</given-names></name> <name><surname>Monroy-de-Jes&#x000FA;s</surname> <given-names>J.</given-names></name> <name><surname>Pacheco-S&#x000E1;nchez</surname> <given-names>J. H.</given-names></name> <name><surname>L&#x000F3;pez-Gonz&#x000E1;lez</surname> <given-names>E.</given-names></name> <name><surname>Antonio-Vel&#x000E1;zquez</surname> <given-names>J. A.</given-names></name></person-group> (<year>2016</year>). <article-title>A selective dynamic sampling back-propagation approach for handling the two-class imbalance problem</article-title>. <source>Appl. Sci.</source> <volume>6</volume>, <fpage>200</fpage>. <pub-id pub-id-type="doi">10.3390/app6070200</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ancona</surname> <given-names>M.</given-names></name> <name><surname>Ceolini</surname> <given-names>E.</given-names></name> <name><surname>&#x000D6;ztireli</surname> <given-names>C.</given-names></name> <name><surname>Gross</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>Gradient-based attribution methods</article-title>, in <source>Explainable AI: Interpreting, Explaining and Visualizing Deep Learning</source>, pp. <fpage>169</fpage>&#x02013;<lpage>191</lpage>. <pub-id pub-id-type="pmid">31974574</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Anderson</surname> <given-names>J. R.</given-names></name> <name><surname>Jones</surname> <given-names>B. W.</given-names></name> <name><surname>Yang</surname> <given-names>J. H.</given-names></name> <name><surname>Shaw</surname> <given-names>M. V.</given-names></name> <name><surname>Watt</surname> <given-names>C. B.</given-names></name> <name><surname>Koshevoy</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>A computational framework for ultrastructural mapping of neural circuitry</article-title>. <source>PLoS Biol</source>. <volume>7</volume>, <fpage>e1000074</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.1000074</pub-id><pub-id pub-id-type="pmid">19855814</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Anderson</surname> <given-names>J. R.</given-names></name> <name><surname>Jones</surname> <given-names>B. W.</given-names></name> <name><surname>Watt</surname> <given-names>C. B.</given-names></name> <name><surname>Shaw</surname> <given-names>M. V.</given-names></name> <name><surname>Yang</surname> <given-names>J. H.</given-names></name> <name><surname>Demill</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Exploring the retinal connectome</article-title>. <source>Mol. Vis.</source> <volume>17</volume>, <fpage>355</fpage>&#x02013;<lpage>379</lpage>. <pub-id pub-id-type="doi">10.7554/eLife.26975</pub-id><pub-id pub-id-type="pmid">21311605</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Arar</surname> <given-names>M.</given-names></name> <name><surname>Ginger</surname> <given-names>Y.</given-names></name> <name><surname>Danon</surname> <given-names>D.</given-names></name> <name><surname>Bermano</surname> <given-names>A. H.</given-names></name> <name><surname>Cohen-Or</surname> <given-names>D.</given-names></name></person-group> (<year>2020</year>). <article-title>Unsupervised Multi-Modal Image Registration via Geometry Preserving Image-to-Image Translation</article-title>, in: <source>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</source>, (<publisher-loc>Seattle, WA</publisher-loc>), <fpage>13407</fpage>&#x02013;<lpage>13416</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arganda-Carreras</surname> <given-names>I.</given-names></name> <name><surname>Turaga</surname> <given-names>S. C.</given-names></name> <name><surname>Berger</surname> <given-names>D. R.</given-names></name> <name><surname>Ciresan</surname> <given-names>D.</given-names></name> <name><surname>Giusti</surname> <given-names>A.</given-names></name> <name><surname>Gambardella</surname> <given-names>L. M.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Crowdsourcing the creation of image segmentation algorithms for connectomics</article-title>. <source>Front. Neuroanat.</source> <volume>9</volume>, <fpage>142</fpage>. <pub-id pub-id-type="doi">10.3389/fnana.2015.00142</pub-id><pub-id pub-id-type="pmid">26594156</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asuni</surname> <given-names>N.</given-names></name> <name><surname>Giachetti</surname> <given-names>A.</given-names></name></person-group> (<year>2008</year>). <article-title>Accuracy improvements and artifacts removal in edge based image interpolation</article-title>. <source>VISAPP</source> <volume>8</volume>, <fpage>58</fpage>&#x02013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.5220/0001074100580065</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bach</surname> <given-names>S.</given-names></name> <name><surname>Binder</surname> <given-names>A.</given-names></name> <name><surname>Montavon</surname> <given-names>G.</given-names></name> <name><surname>Klauschen</surname> <given-names>F.</given-names></name> <name><surname>Muller</surname> <given-names>K. R.</given-names></name> <name><surname>Samek</surname> <given-names>W.</given-names></name></person-group> (<year>2015</year>). <article-title>On pixel-wise explanations for non-linear classifier decisions by layer-wise relevance propagation</article-title>. <source>PLoS ONE</source> <volume>10</volume>, <fpage>e0130140</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0130140</pub-id><pub-id pub-id-type="pmid">26161953</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chaurasia</surname> <given-names>A.</given-names></name> <name><surname>Culurciello</surname> <given-names>E.</given-names></name></person-group> (<year>2017</year>). <article-title>Linknet: Exploiting encoder representations for efficient semantic segmentation</article-title>, in <source>2017 IEEE Visual Communications and Image Processing (VCIP)</source>, <fpage>1</fpage>&#x02013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.-C.</given-names></name> <name><surname>Papandreou</surname> <given-names>G.</given-names></name> <name><surname>Kokkinos</surname> <given-names>I.</given-names></name> <name><surname>Murphy</surname> <given-names>K.</given-names></name> <name><surname>Yuille</surname> <given-names>A. L.</given-names></name></person-group> (<year>2014</year>). <article-title>Semantic image segmentation with deep convolutional nets and fully connected crfs</article-title>. <source>arXiv [Preprint]. arXiv:</source>1412.7062. <pub-id pub-id-type="doi">10.48550/arXiv.1412.7062</pub-id><pub-id pub-id-type="pmid">28463186</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Cui</surname> <given-names>K.</given-names></name> <name><surname>Qian</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>Collaborative global-local networks for memory-efficient segmentation of ultra-high resolution images</article-title>, in: <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, p. <fpage>8924</fpage>&#x02013;<lpage>8933</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>K.</given-names></name> <name><surname>Khandelwal</surname> <given-names>U.</given-names></name> <name><surname>Levy</surname> <given-names>O.</given-names></name> <name><surname>Manning</surname> <given-names>C. D.</given-names></name></person-group> (<year>2019</year>). <article-title>What does bert look at? an analysis of bert&#x00027;s attention</article-title>. <source>arXiv [Preprint]. arXiv:</source>1906.04341. <pub-id pub-id-type="doi">10.18653/v1/W19-4828</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Conrad</surname> <given-names>R.</given-names></name> <name><surname>Narayan</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>CEM500K, a large-scale heterogeneous unlsabeled cellular electron microscopy image dataset for deep learning</article-title>. <source>Elife</source> <volume>10</volume>, <fpage>65894</fpage>. <pub-id pub-id-type="doi">10.7554/eLife.65894</pub-id><pub-id pub-id-type="pmid">33830015</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>Y.</given-names></name> <name><surname>Jia</surname> <given-names>M.</given-names></name> <name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Class-balanced loss based on effective number of samples</article-title>, in: <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, p. <fpage>9268</fpage>&#x02013;<lpage>9277</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>DeBello</surname> <given-names>W. M.</given-names></name> <name><surname>McBride</surname> <given-names>T. J.</given-names></name> <name><surname>Nichols</surname> <given-names>G. S.</given-names></name> <name><surname>Pannoni</surname> <given-names>K. E.</given-names></name> <name><surname>Sanculi</surname> <given-names>D.</given-names></name> <name><surname>Totten</surname> <given-names>D. J.</given-names></name></person-group> (<year>2014</year>). <article-title>Input clustering and the microscale structure of local circuits</article-title>. <source>Front. Neural. Circuits.</source> <volume>8</volume>, <fpage>112</fpage>. <pub-id pub-id-type="doi">10.3389/fncir.2014.00112</pub-id><pub-id pub-id-type="pmid">25309336</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dice</surname> <given-names>L. R.</given-names></name></person-group> (<year>1945</year>). <article-title>Measures of the amount of ecologic association between species</article-title>. <source>Ecology</source> <volume>26</volume>, <fpage>297</fpage>&#x02013;<lpage>302</lpage>. <pub-id pub-id-type="doi">10.2307/1932409</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fern&#x000E1;ndez</surname> <given-names>A.</given-names></name> <name><surname>Garcia</surname> <given-names>S.</given-names></name> <name><surname>Herrera</surname> <given-names>F.</given-names></name> <name><surname>Chawla</surname> <given-names>N. V.</given-names></name></person-group> (<year>2018</year>). <article-title>SMOTE for learning from imbalanced data: progress and challenges, marking the 15-year anniversary</article-title>. <source>J. Artific. Intell. Res.</source> <volume>61</volume>, <fpage>863</fpage>&#x02013;<lpage>905</lpage>. <pub-id pub-id-type="doi">10.1613/jair.1.11192</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Goceri</surname> <given-names>E.</given-names></name></person-group> (<year>2019</year>). <article-title>Challenges and recent solutions for image segmentation in the era of deep learning</article-title>, in: <source>International Conference on Image Processing Theory, Tools and Applications</source>, p. <fpage>1</fpage>&#x02013;<lpage>6</lpage>. 087</citation>
</ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Golgi</surname> <given-names>C.</given-names></name></person-group> (<year>1885</year>). <source>Sulla fina anatomia degli organi centrali del sistema nervoso</source>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guidotti</surname> <given-names>R.</given-names></name> <name><surname>Monreale</surname> <given-names>A.</given-names></name> <name><surname>Ruggieri</surname> <given-names>S.</given-names></name> <name><surname>Turini</surname> <given-names>F.</given-names></name> <name><surname>Giannotti</surname> <given-names>F.</given-names></name> <name><surname>Pedreschi</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>A survey of methods for explaining black box models</article-title>. <source>ACM Comput. Surv.</source> <volume>51</volume>, <fpage>1</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1145/3236009</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hariharan</surname> <given-names>B.</given-names></name> <name><surname>Arbel&#x000E1;ez</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Malik</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>Hypercolumns for object segmentation and fine-grained localization</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, p. <fpage>447</fpage>&#x02013;<lpage>456</lpage>. <pub-id pub-id-type="pmid">27295654</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Deep residual learning for image recognition</article-title>, in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, p. <fpage>770</fpage>&#x02013;<lpage>778</lpage>. <pub-id pub-id-type="pmid">32166560</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heimann</surname> <given-names>T.</given-names></name> <name><surname>van Ginneken</surname> <given-names>B.</given-names></name> <name><surname>Styner</surname> <given-names>M. A.</given-names></name> <name><surname>Arzhaeva</surname> <given-names>Y.</given-names></name> <name><surname>Aurich</surname> <given-names>V.</given-names></name> <name><surname>Bauer</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Comparison and evaluation of methods for liver segmentation from CT datasets</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>28</volume>, <fpage>1251</fpage>&#x02013;<lpage>1265</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2009.2013851</pub-id><pub-id pub-id-type="pmid">19211338</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hwang</surname> <given-names>J. W.</given-names></name> <name><surname>Lee</surname> <given-names>H. S.</given-names></name></person-group> (<year>2004</year>). <article-title>Adaptive image interpolation based on local gradient features</article-title>. <source>IEEE Signal Process. Lett.</source> <volume>11</volume>, <fpage>359</fpage>&#x02013;<lpage>362</lpage>. <pub-id pub-id-type="doi">10.1109/LSP.2003.821718</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>Adam: A method for stochastic optimization</article-title>. <source>arXiv [Preprint]. arXiv:</source>1412.6980. <pub-id pub-id-type="doi">10.48550/arXiv.1412.6980</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kornfeld</surname> <given-names>J.</given-names></name> <name><surname>Denk</surname> <given-names>W.</given-names></name></person-group> (<year>2018</year>). <article-title>Progress and remaining challenges in high-throughput volume electron microscopy</article-title>. <source>Curr. Opin. Neurobiol.</source> <volume>50</volume>, <fpage>261</fpage>&#x02013;<lpage>267</lpage>. <pub-id pub-id-type="doi">10.1016/j.conb.2018.04.030</pub-id><pub-id pub-id-type="pmid">31160001</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). <article-title>Imagenet classification with deep convolutional neural networks</article-title>. <source>Adv. Neural Inform. Process. Syst.</source> <volume>25</volume>, <fpage>1097</fpage>&#x02013;<lpage>1105</lpage>. <pub-id pub-id-type="doi">10.1145/3065386</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leitch</surname> <given-names>B.</given-names></name></person-group> (<year>1992</year>). <article-title>Ultrastructure of electrical synapses: review</article-title>. <source>Electron. Microsc. Rev.</source> <volume>5</volume>, <fpage>311</fpage>&#x02013;<lpage>339</lpage>. <pub-id pub-id-type="doi">10.1016/0892-0354(92)90014-H</pub-id><pub-id pub-id-type="pmid">1581553</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>D. C.</given-names></name> <name><surname>Liu</surname> <given-names>C. W.</given-names></name> <name><surname>Hu</surname> <given-names>S. C.</given-names></name></person-group> (<year>2010</year>). <article-title>A learning method for the class imbalance problem with medical data sets</article-title>. <source>Comput. Biol. Med.</source> <volume>40</volume>, <fpage>509</fpage>&#x02013;<lpage>518</lpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2010.03.005</pub-id><pub-id pub-id-type="pmid">20347072</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Goyal</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>Focal loss for dense object</article-title>. <source>Detection</source> <volume>12</volume>, <fpage>2980</fpage>&#x02013;<lpage>2988</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.324</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>W.</given-names></name> <name><surname>Dong</surname> <given-names>L.</given-names></name></person-group> (<year>2006</year>). <article-title>Adaptive downsampling to improve image compression at low bit rates</article-title>. <source>IEEE Trans. Image Process.</source> <volume>15</volume>, <fpage>2513</fpage>&#x02013;<lpage>2521</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2006.877415</pub-id><pub-id pub-id-type="pmid">16948298</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>M.</given-names></name> <name><surname>Qu</surname> <given-names>Q.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Generative adversarial network for abstractive text summarization</article-title>, in: <source>Proceedings of the AAAI Conference on Artificial Intelligence</source>, p. <fpage>10</fpage>. <pub-id pub-id-type="pmid">32701451</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Deng</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>A Unified Framework of Surrogate Loss by Refactoring and Interpolation</article-title>, in <source>Computer Vision &#x02013; ECCV 2020</source>, eds. <person-group person-group-type="editor"><name><surname>Vedaldi</surname> <given-names>A.</given-names></name> <name><surname>Bischof</surname> <given-names>H.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name> <name><surname>Frahm</surname> <given-names>J.-M.</given-names></name> <name><surname>Vedaldi</surname> <given-names>A.</given-names></name> <name><surname>Bischof</surname> <given-names>H.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name> <name><surname>Frahm</surname> <given-names>J.-M.</given-names></name></person-group> (<publisher-loc>Springer: Cham</publisher-loc>), p. <fpage>278</fpage>&#x02013;<lpage>293</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mastronarde</surname> <given-names>D. N.</given-names></name></person-group> (<year>2005</year>). <article-title>Automated electron microscope tomography using robust prediction of specimen movements</article-title>. <source>J. Struct. Biol.</source> <volume>152</volume>, <fpage>36</fpage>&#x02013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1016/j.jsb.2005.07.007</pub-id><pub-id pub-id-type="pmid">16182563</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montavon</surname> <given-names>G.</given-names></name> <name><surname>Lapuschkin</surname> <given-names>S.</given-names></name> <name><surname>Binder</surname> <given-names>A.</given-names></name> <name><surname>Samek</surname> <given-names>W.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>K.-R.</given-names></name></person-group> (<year>2017</year>). <article-title>Explaining nonlinear classification decisions with deep taylor decomposition</article-title>. <source>Pattern Recogn.</source> <volume>65</volume>, <fpage>211</fpage>&#x02013;<lpage>222</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2016.11.008</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ronneberger</surname> <given-names>O.</given-names></name> <name><surname>Fischer</surname> <given-names>P.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). <article-title>U-Net: Convolutional Networks for Biomedical Image Segmentation</article-title>, in: <source>International Conference on Medical image computing and computer-assisted intervention</source>, p. <fpage>28</fpage></citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>A. G.</given-names></name> <name><surname>Navab</surname> <given-names>N.</given-names></name> <name><surname>Wachinger</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>Concurrent spatial and channel &#x02018;squeeze and excitation&#x02019;in fully convolutional networks</article-title>, in: <source>International conference on medical image computing and computer-assisted intervention</source>, p. <fpage>421</fpage>&#x02013;<lpage>429</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Russakovsky</surname> <given-names>O.</given-names></name> <name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Su</surname> <given-names>H.</given-names></name> <name><surname>Krause</surname> <given-names>J.</given-names></name> <name><surname>Satheesh</surname> <given-names>S.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Imagenet large scale visual recognition challenge</article-title>. <source>Int. J. Comput. Vision</source> <volume>115</volume>, <fpage>211</fpage>&#x02013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-015-0816-y</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saltelli</surname> <given-names>A.</given-names></name> <name><surname>Annoni</surname> <given-names>P.</given-names></name> <name><surname>Azzini</surname> <given-names>I.</given-names></name> <name><surname>Campolongo</surname> <given-names>F.</given-names></name> <name><surname>Ratto</surname> <given-names>M.</given-names></name> <name><surname>Tarantola</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Variance based sensitivity analysis of model output</article-title>. <source>Design Estimat. Total Sensitiv. Index. Comput. Physics Commun.</source> <volume>181</volume>, <fpage>259</fpage>&#x02013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.1016/j.cpc.2009.09.018</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Sasaki</surname> <given-names>Y.</given-names></name> <name><surname>Fellow</surname> <given-names>R.</given-names></name></person-group> (<year>2007</year>). <source>The Truth of the F-Measure</source>. <publisher-loc>Manchester</publisher-loc>: <publisher-name>MIB-School of Computer Science, University of Manchester</publisher-name>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cs.odu.edu/&#x0007E;mukka/cs795sum09dm/Lecturenotes/Day3/F-measure-YS-26Oct07.pdf">https://www.cs.odu.edu/&#x0007E;mukka/cs795sum09dm/Lecturenotes/Day3/F-measure-YS-26Oct07.pdf</ext-link></citation>
</ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Schulz</surname> <given-names>K.</given-names></name> <name><surname>Leon</surname> <given-names>S.</given-names></name> <name><surname>Federico</surname> <given-names>T.</given-names></name> <name><surname>Tim</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>Restricting the Flow: Information Bottlenecks for Attribution</article-title>. <source>arXiv [Preprint]. arXiv:2001.00396</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2001.00396</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Shapson-Coe</surname> <given-names>A.</given-names></name> <name><surname>Januszewski</surname> <given-names>M.</given-names></name> <name><surname>Berger</surname> <given-names>D. R.</given-names></name> <name><surname>Pope</surname> <given-names>A.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>A connectomic study of a petascale fragment of human cerebral cortex</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2021.05.29.446289</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shawn</surname> <given-names>M.</given-names></name></person-group> (<year>2016</year>). <article-title>Progress towards mammalian whole-brain cellular connectomics</article-title>. <source>Front. Neuroanatom.</source> <volume>10</volume>, <fpage>62</fpage>. <pub-id pub-id-type="doi">10.3389/fnana.2016.00062</pub-id><pub-id pub-id-type="pmid">27445704</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv [Preprint]. arXiv:</source>1409.1556. <pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Subbarao</surname> <given-names>M.</given-names></name> <name><surname>Tyan</surname> <given-names>J. K.</given-names></name></person-group> (<year>1998</year>). <article-title>Selecting the optimal focus measure for autofocusing and depth-from-focus</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intel.</source> <volume>20</volume>, <fpage>864</fpage>&#x02013;<lpage>870</lpage>. <pub-id pub-id-type="doi">10.1109/34.709612</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sundararajan</surname> <given-names>M.</given-names></name> <name><surname>Taly</surname> <given-names>A.</given-names></name> <name><surname>Yan</surname> <given-names>Q.</given-names></name></person-group> (<year>2017</year>). <article-title>Axiomatic attribution for deep networks</article-title>. <source>arXiv [Preprint]. arXiv:1703.01365</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1703.01365</pub-id><pub-id pub-id-type="pmid">33265480</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Sermanet</surname> <given-names>P.</given-names></name> <name><surname>Reed</surname> <given-names>S.</given-names></name> <name><surname>Anguelov</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Going deeper with convolutions</article-title>, in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>), <fpage>1</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298594</pub-id></citation>
</ref>
<ref id="B49">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>M.</given-names></name> <name><surname>Le</surname> <given-names>Q.</given-names></name></person-group> (<year>2019</year>). <article-title>Efficientnet: Rethinking model scaling for convolutional neural networks</article-title>, in <source>International conference on machine learning</source>, p. <fpage>6105</fpage>&#x02013;<lpage>6114</lpage>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thakkinstian</surname> <given-names>A.</given-names></name> <name><surname>McElduff</surname> <given-names>P.</given-names></name> <name><surname>D&#x00027;Este</surname> <given-names>C.</given-names></name> <name><surname>Duffy</surname> <given-names>D.</given-names></name> <name><surname>Attia</surname> <given-names>J.</given-names></name></person-group> (<year>2005</year>). <article-title>A method for meta-analysis of molecular association studies</article-title>. <source>Stat. Med.</source> <volume>24</volume>, <fpage>1291</fpage>&#x02013;<lpage>1306</lpage>. <pub-id pub-id-type="doi">10.1002/sim.2010</pub-id><pub-id pub-id-type="pmid">15568190</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vollath</surname> <given-names>D.</given-names></name></person-group> (<year>2008</year>). <article-title>Nanomaterials an introduction to synthesis, properties and application</article-title>. <source>Environ. Eng. Manage. J.</source> <volume>7</volume>, <fpage>865</fpage>&#x02013;<lpage>870</lpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Vu</surname> <given-names>T.</given-names></name> <name><surname>Van Nguyen</surname> <given-names>C.</given-names></name> <name><surname>Pham</surname> <given-names>T. X.</given-names></name> <name><surname>Luu</surname> <given-names>T. M.</given-names></name> <name><surname>Yoo</surname> <given-names>C. D.</given-names></name></person-group> (<year>2018</year>). <article-title>Fast and efficient image quality enhancement via desubpixel convolutional neural networks</article-title>, in <source>Proceedings of the European Conference on Computer Vision (ECCV) Workshops</source>, <fpage>12</fpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Bovik</surname> <given-names>A. C.</given-names></name></person-group> (<year>2006</year>). <article-title>Modern image quality assessment</article-title>. <source>Synthesis Lect Image, Video Multimedia Process.</source> <volume>2</volume>, <fpage>1</fpage>&#x02013;<lpage>156</lpage>. <pub-id pub-id-type="doi">10.2200/S00010ED1V01Y200508IVM003</pub-id></citation>
</ref>
<ref id="B54">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Simoncelli</surname> <given-names>E. P.</given-names></name> <name><surname>Bovik</surname> <given-names>A. C.</given-names></name></person-group> (<year>2003</year>). <article-title>Multiscale structural similarity for image quality assessment</article-title>, in <source>The Thrity-Seventh Asilomar Conference on Signals, Systems and Computers</source>, p. <fpage>1398</fpage>&#x02013;<lpage>1402</lpage>.</citation>
</ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>White</surname> <given-names>J. G.</given-names></name> <name><surname>Southgate</surname> <given-names>E.</given-names></name> <name><surname>Thomson</surname> <given-names>J. N.</given-names></name> <name><surname>Brenner</surname> <given-names>S.</given-names></name></person-group> (<year>1986</year>). <article-title>The structure of the nervous system of the nematode <italic>Caenorhabditis elegans</italic></article-title>. <source>Philos. Trans. R Soc. Lond. B Biol. Sci.</source> <volume>314</volume>, <fpage>1</fpage>&#x02013;<lpage>340</lpage>. <pub-id pub-id-type="doi">10.1098/rstb.1986.0056</pub-id><pub-id pub-id-type="pmid">25750233</pub-id></citation></ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>S.</given-names></name> <name><surname>Wen</surname> <given-names>W.</given-names></name> <name><surname>Xiao</surname> <given-names>B.</given-names></name> <name><surname>Guo</surname> <given-names>X.</given-names></name> <name><surname>Du</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>An accurate skeleton extraction approach from 3d point clouds of maize plants</article-title>. <source>Front Plant. Sci.</source> <volume>10</volume>, <fpage>248</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2019.00248</pub-id><pub-id pub-id-type="pmid">30899271</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>y Cajal</surname> <given-names>S. R.</given-names></name></person-group> (<year>1888</year>). <source>Estructura de los centros nerviosos de las aves</source>.</citation>
</ref>
<ref id="B58">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yoo</surname> <given-names>J.</given-names></name> <name><surname>Ahn</surname> <given-names>N.</given-names></name> <name><surname>Sohn</surname> <given-names>K.-A.</given-names></name></person-group> (<year>2020</year>). <article-title>Rethinking data augmentation for image super-resolution: a comprehensive analysis and a new strategy</article-title>. <source>arXiv:2004.00448 [cs, eess]</source>.</citation>
</ref>
<ref id="B59">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Feng</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>M.-Y.</given-names></name> <name><surname>Ramalingam</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>Casenet: Deep category-aware semantic edge detection</article-title>, in: <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, p. <fpage>5964</fpage>&#x02013;<lpage>5973</lpage>.</citation>
</ref>
<ref id="B60">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Su</surname> <given-names>C.</given-names></name> <name><surname>Zheng</surname> <given-names>L.</given-names></name> <name><surname>Xie</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>Correlating Edge, Pose With Parsing</article-title>, in: <source>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</source>, (<publisher-loc>WA: Seattle</publisher-loc>), p. <fpage>8897</fpage>&#x02013;<lpage>8906</lpage>.</citation>
</ref>
<ref id="B61">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>You</surname> <given-names>A.</given-names></name> <name><surname>Bai</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>Semantically multi-modal image synthesis</article-title>, in: <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, p. <fpage>5467</fpage>&#x02013;<lpage>5476</lpage>.</citation>
</ref>
</ref-list>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link ext-link-type="uri" xlink:href="https://www.biendata.xyz/competition/urisc/">https://www.biendata.xyz/competition/urisc/</ext-link></p></fn>
</fn-group>
</back>
</article>