<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cell. Neurosci.</journal-id>
<journal-title>Frontiers in Cellular Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cell. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-5102</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fncel.2023.1127847</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>NeuroSeg-II: A deep learning approach for generalized neuron segmentation in two-photon Ca<sup>2+</sup> imaging</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Xu</surname> <given-names>Zhehao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2151283/overview"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Wu</surname> <given-names>Yukun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2214199/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Guan</surname> <given-names>Jiangheng</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1093666/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liang</surname> <given-names>Shanshan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/551867/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Pan</surname> <given-names>Junxia</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Meng</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/551869/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Hu</surname> <given-names>Qianshuo</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Jia</surname> <given-names>Hongbo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1474549/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Chen</surname> <given-names>Xiaowei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/363377/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Liao</surname> <given-names>Xiang</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/38104/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Advanced Institute for Brain and Intelligence, Medical College, Guangxi University</institution>, <addr-line>Nanning</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Neurosurgery, The General Hospital of Chinese PLA Central Theater Command</institution>, <addr-line>Wuhan</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Brain Research Center and State Key Laboratory of Trauma, Burns, and Combined Injury, Third Military Medical University</institution>, <addr-line>Chongqing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Center for Neurointelligence, School of Medicine, Chongqing University</institution>, <addr-line>Chongqing</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>School of Artificial Intelligence, Chongqing University of Technology</institution>, <addr-line>Chongqing</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>Brain Research Instrument Innovation Center, Suzhou Institute of Biomedical Engineering and Technology, Chinese Academy of Sciences</institution>, <addr-line>Suzhou, Jiangsu</addr-line>, <country>China</country></aff>
<aff id="aff7"><sup>7</sup><institution>Guangyang Bay Laboratory, Chongqing Institute for Brain and Intelligence</institution>, <addr-line>Chongqing</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Jiamin Wu, Tsinghua University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Peixian Zhuang, Tsinghua University, China; Wan Sen, Tsinghua University, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Xiaowei Chen, <email>xiaowei_chen@tmmu.edu.cn</email></corresp>
<corresp id="c002">Xiang Liao, <email>xiang.liao@cqu.edu.cn</email></corresp>
<fn fn-type="equal" id="fn002"><p><sup>&#x2020;</sup>These authors have contributed equally to this work</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Cellular Neurophysiology, a section of the journal Frontiers in Cellular Neuroscience</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>06</day>
<month>04</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>17</volume>
<elocation-id>1127847</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>03</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2023 Xu, Wu, Guan, Liang, Pan, Wang, Hu, Jia, Chen and Liao.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Xu, Wu, Guan, Liang, Pan, Wang, Hu, Jia, Chen and Liao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>The development of two-photon microscopy and Ca<sup>2+</sup> indicators has enabled the recording of multiscale neuronal activities <italic>in vivo</italic> and thus advanced the understanding of brain functions. However, it is challenging to perform automatic, accurate, and generalized neuron segmentation when processing a large amount of imaging data. Here, we propose a novel deep-learning-based neural network, termed as NeuroSeg-II, to conduct automatic neuron segmentation for <italic>in vivo</italic> two-photon Ca<sup>2+</sup> imaging data. This network architecture is based on Mask region-based convolutional neural network (R-CNN) but has enhancements of an attention mechanism and modified feature hierarchy modules. We added an attention mechanism module to focus the computation on neuron regions in imaging data. We also enhanced the feature hierarchy to extract feature information at diverse levels. To incorporate both spatial and temporal information in our data processing, we fused the images from average projection and correlation map extracting the temporal information of active neurons, and the integrated information was expressed as two-dimensional (2D) images. To achieve a generalized neuron segmentation, we conducted a hybrid learning strategy by training our model with imaging data from different labs, including multiscale data with different Ca<sup>2+</sup> indicators. The results showed that our approach achieved promising segmentation performance across different imaging scales and Ca<sup>2+</sup> indicators, even including the challenging data of large field-of-view mesoscopic images. By comparing state-of-the-art neuron segmentation methods for two-photon Ca<sup>2+</sup> imaging data, we showed that our approach achieved the highest accuracy with a publicly available dataset. Thus, NeuroSeg-II enables good segmentation accuracy and a convenient training and testing process.</p>
</abstract>
<kwd-group>
<kwd>two-photon Ca<sup>2+</sup> imaging</kwd>
<kwd>generalized neuron segmentation</kwd>
<kwd>deep learning</kwd>
<kwd>attention mechanism</kwd>
<kwd>hybrid training</kwd>
</kwd-group>
<contract-num rid="cn001">32171096</contract-num>
<contract-num rid="cn001">31925018</contract-num>
<contract-num rid="cn001">32127801</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<counts>
<fig-count count="8"/>
<table-count count="0"/>
<equation-count count="7"/>
<ref-count count="51"/>
<page-count count="16"/>
<word-count count="10249"/>
</counts>
</article-meta>
</front>
<body>
 <sec id="S1" sec-type="intro">
<title>1. Introduction</title>
<p>The fast advances in two-photon microscopy (<xref ref-type="bibr" rid="B17">Helmchen and Denk, 2005</xref>; <xref ref-type="bibr" rid="B12">Grewe et al., 2010</xref>; <xref ref-type="bibr" rid="B45">Stringer et al., 2019</xref>) and various Ca<sup>2+</sup> indicators (<xref ref-type="bibr" rid="B1">Akerboom et al., 2013</xref>; <xref ref-type="bibr" rid="B7">Chen et al., 2013</xref>; <xref ref-type="bibr" rid="B8">Dana et al., 2019</xref>) have enabled researchers to record individual neurons <italic>in vivo</italic> at a large scale and high speed. Experiments have been performed to study brain functions with activities from many neurons in targeted brain regions. Accurately segmenting neurons carrying biological information is an essential step for analyzing the spatiotemporal data recorded by functional imaging experiments. Manual neuron segmentation is accurate and can screen out regions with unnecessary information, and thus, the manual segmentation result is defined as ground truth (GT). However, owing to the increasing amount of data (<xref ref-type="bibr" rid="B14">Harris et al., 2016</xref>) generated by the large size of the imaging field and the number of recorded neurons (<xref ref-type="bibr" rid="B22">Kim and Schnitzer, 2022</xref>), human annotators encounter a considerable workload. In addition, different annotators have their specific neuron labeling criteria, which may generate inconsistent results.</p>
<p>In the last decade, neuron segmentation approaches have continuously advanced in accuracy and computational speed. Currently, the methods can complete processing with a speed far exceeding that of human annotators and provide segmentation accuracy that is close to that of human annotators (<xref ref-type="bibr" rid="B33">Pnevmatikakis, 2019</xref>; <xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>). The neuron segmentation methods are currently divided into two categories: unsupervised and supervised methods. For the first category of neuron segmentation methods (unsupervised), they typically identify pixels representing a neuronal structure and integrate these pixels into a region of neurons by intensity. This type of algorithm normally segments neurons with component analysis, including principal component analysis or independent component analysis (PCA/ICA) (<xref ref-type="bibr" rid="B29">Mukamel et al., 2009</xref>), non-negative matrix factorization (NMF) (<xref ref-type="bibr" rid="B28">Maruyama et al., 2014</xref>) and constrained non-negative matrix factorization (CNMF) (<xref ref-type="bibr" rid="B34">Pnevmatikakis et al., 2016</xref>), or the activity model (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>). For example, Suite2p (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>) uses spatial region of interest (ROI) shapes and neuronal activity traces from imaging data to segment neurons.</p>
<p>For the second category of neuron segmentation methods (supervised), they are trained to extract neuron features from labeled two-dimensional (2D) data (images) or three-dimensional (3D) data (videos). Convolutional neural networks (CNNs) are typically designed for supervised learning. Based on different types of data processing, CNNs can be divided into 2D CNN and 3D CNN. 2D CNN extracts features with training on manually labeled masks in image. For example, Mask region-based convolutional neural network (R-CNN) (<xref ref-type="bibr" rid="B15">He et al., 2017</xref>) is an instance segmentation algorithm and it can be applied for segmenting neuron in an image. In contrast to 2D CNN, 3D CNN is trained to extract features from labeled video data. For example, STNeuroNet (<xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>) was proposed to use a 3D CNN for neuron segmentation and exploit the spatiotemporal information in two-photon Ca<sup>2+</sup> imaging data. Shallow U-Net Neuron Segmentation (SUNS) (<xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>) uses shallow CNN with U-shaped architecture to extract the spatial features of neurons, realizing fast and accurate segmentation. The 2D CNN and 3D CNN have their specific advantages and limitations. Neuron segmentation algorithms with 2D CNN are flexible and fast (<xref ref-type="bibr" rid="B43">Stoyanov et al., 2018</xref>). For this class of methods, the images of the training dataset can cover various Ca<sup>2+</sup> indicators, imaging scales and imaging depths, which help this class of methods achieve a certain extent of generalizability and robustness. However, the temporal information of imaging data is lost when the video data are converted into an image. By contrast, neuron segmentation methods with 3D CNN can capture the temporal information of neurons (<xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>), particularly they can help recognize overlapping neurons. However, this class of methods requires long recording and highly active neurons. In addition, some recent methods, e.g., CaImAn (<xref ref-type="bibr" rid="B11">Giovannucci et al., 2019</xref>), combine these two kinds of machine learning algorithms by identifying activity components using unsupervised learning and evaluating these components using supervised learning. However, it is still challenging for the existing methods to perform generalized neuron segmentation with complex two-photon Ca<sup>2+</sup> imaging data, so we aim to develop a method that can accurately segment neurons in various situations.</p>
<p>In our previous study (<xref ref-type="bibr" rid="B13">Guan et al., 2018</xref>; <xref ref-type="bibr" rid="B36">Shen et al., 2018</xref>), NeuroSeg was developed to achieve unsupervised neuron segmentation for <italic>in vivo</italic> two-photon Ca<sup>2+</sup> imaging data by using a generalized Laplacian of Gaussian filter to detect neurons and weighting-based segmentation to separate individual neurons. However, its model has the limitation of performing neuron segmentation in two-photon Ca<sup>2+</sup> imaging data of different Ca<sup>2+</sup> indicators. Hence this method demands further development. To segment both active and inactive neurons in imaging data across Ca<sup>2+</sup> indicators, imaging scales, brain regions and imaging depths, here we propose NeuroSeg-II, a deep learning model based on an attention mechanism and enhanced feature hierarchy, to perform neuron segmentation in two-photon Ca<sup>2+</sup> imaging with a 2D image processing approach. As the sparsely firing neurons may be hardly visible in the average or maximum projected images, the correlation map can make these neurons visible (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>). In preprocessing, we fused the average image with correlation map to integrate the spatial and temporal information and generate a new 2D image for neuron segmentation. To train and validate NeuroSeg-II&#x2019;s performance, we used the datasets acquired from our lab and publicly available datasets. The results show that NeuroSeg-II solved the problem of generalized neuron segmentation in two-photon Ca<sup>2+</sup> imaging data and achieved good performance across different Ca<sup>2+</sup> indicators (OGB-1, Cal-520, and GCaMP6), multiple imaging scales, different brain regions and imaging depths. NeuroSeg-II used spatiotemporal activity information with fused images and successfully segmented active and inactive neurons, indicating that it has good generalizability for processing different types of imaging data. By comparing the other methods for neuron segmentation with the publicly available two-photon Ca<sup>2+</sup> imaging dataset, we found that our approach outperformed other competitors in accuracy. Therefore, our deep learning approach is efficient in performing generalized neuron segmentation in two-photon Ca<sup>2+</sup> imaging, which is complementary to NeuroSeg and may facilitate future neuroscience research.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>2. Materials and methods</title>
<sec id="S2.SS1">
<title>2.1. Functional two-photon imaging datasets</title>
<sec id="S2.SS1.SSS1">
<title>2.1.1. Data acquisition in our lab</title>
<p>In this study, C57BL/6J mice (2&#x2013;3 months old) were used for two-photon Ca<sup>2+</sup> imaging experiments. The mice were provided by the Laboratory Animal Center at the Third Military Medical University, and the experimental procedures were performed based on protocols approved by the Third Military Medical University Animal Care and Use Committee.</p>
<p>The two-photon Ca<sup>2+</sup> imaging experiments were conducted in the mouse auditory cortex (<xref ref-type="bibr" rid="B24">Li et al., 2017</xref>; <xref ref-type="bibr" rid="B47">Wang M. et al., 2020</xref>). After we anesthetized the mouse with isoflurane, the mouse&#x2019;s skull was glued with a prefabricated plastic chamber. The auditory cortex region was exposed by a small craniotomy (&#x223C;4 mm<sup>2</sup>) and injected with indicator (OGB-1 AM, Cal-520 AM, or GCaMP6f). After 2 h, Ca<sup>2+</sup> imaging was performed with a mode-locked Ti:Sa laser (Mai-Tai DeepSee, Spectra Physics, Santa Clara, CA, USA) delivering two-photon excitation light. A custom-built two-photon microscope system (LotosScan, Suzhou Institute of Biomedical Engineering and Technology, Suzhou, China) was used to record the imaging data (<xref ref-type="bibr" rid="B20">Jia et al., 2010</xref>, <xref ref-type="bibr" rid="B21">2014</xref>).</p>
<p>This dataset consisted of 193 two-photon Ca<sup>2+</sup> imaging videos, comprising 61 data samples for expressing the OGB-1 indicator, 127 data samples for expressing the Cal-520 indicator, and five data samples for expressing the GCaMP6f indicator. Three experienced annotators labeled each neuron independently and then compared their labeled results to produce a final consensus as the GT.</p>
</sec>
<sec id="S2.SS1.SSS2">
<title>2.1.2. Allen brain observatory (ABO) dataset</title>
<p>The ABO dataset consists of neuronal population imaging across different brain regions and layers with two-photon microscopy. We used the ABO dataset consisting of 132 images from ALLEN BRAIN ATLAS (Sessions A&#x2013;C). This image dataset includes six images recorded at a depth of 175 &#x03BC;m in the rostrolateral visual cortex (VISrl), 12 images recorded at a depth of 175 &#x03BC;m in the posterolateral visual cortex (VISpm), nine images recorded at a depth of 275 &#x03BC;m in the VISpm, 16 images recorded at a depth of 175 &#x03BC;m in the VISp, 25 images recorded at a depth of 275 &#x03BC;m in the VISp, 12 images recorded at a depth of 175 &#x03BC;m in the lateral visual cortex (VISl), 15 images recorded at a depth of 275 &#x03BC;m in the VISl, six images recorded at a depth of 175 &#x03BC;m in the anteromedial visual cortex (VISam), 13 images recorded at a depth of 175 &#x03BC;m in the anterolateral visual cortex (VISal), and 18 images recorded at a depth of 275 &#x03BC;m in the VISal. All mice in the above experiments expressed the GCaMP6f indicator. Three experienced annotators labeled each neuron independently and then compared their labeled results to produce a final consensus as GT.</p>
</sec>
<sec id="S2.SS1.SSS3">
<title>2.1.3. Neurofinder challenge dataset</title>
<p>The Neurofinder dataset consists of neuronal population imaging across different brain regions with two-photon microscopy. The dataset was annotated in three different laboratories, resulting in diverse sub-datasets. We used 10 imaging data (videos) samples from this dataset. All mice expressed the GCaMP6s indicator. We used the neuron GT from the work of STNeuroNet (<xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>). Each group of videos contained one training data sample and one testing data sample. To increase the number of images and improve image quality, we used the preprocessing method to convert each video dataset into seven corresponding images (six images from an evenly divided video and one image from the whole video).</p>
</sec>
<sec id="S2.SS1.SSS4">
<title>2.1.4. Large-field mesoscopic two-photon imaging dataset</title>
<p>A single image data sample of mesoscopic large-field imaging was recorded with a mouse expressing GCaMP6s under the thy-1 promoter (<xref ref-type="bibr" rid="B40">Sofroniew et al., 2016</xref>). As the dimensions of the image data are too large (1,792 pixels &#x00D7; 1,682 pixels) to fit the neural network, the image data cannot be tested directly. Thus, we segmented the original image into small images for testing. The neuron GT is provided with this dataset.</p>
</sec>
</sec>
<sec id="S2.SS2">
<title>2.2. Image preprocessing</title>
<p>First, the motion corrected imaging data (videos) were converted into images by average projection (<xref ref-type="bibr" rid="B43">Stoyanov et al., 2018</xref>) and correlation map (<xref ref-type="bibr" rid="B10">Foroosh et al., 2002</xref>; <xref ref-type="bibr" rid="B2">Alba et al., 2015</xref>) to represent the spatiotemporal information of neurons. To obtain the correlation map, we calculated the multidimensional correlation of each pixel and its surrounding pixels to localize the neurons. Here, we calculated a weighted multidimensional correlation (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>) as</p>
<disp-formula id="S2.E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">&#x22EF;</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mo fence="true">||</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mo largeop="true" symmetric="true">&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo fence="true">||</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:msub>
<mml:mo largeop="true" symmetric="true">&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo fence="true">||</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo fence="true">||</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>c<sub>w</sub></italic> is the calculation at each pixel for different dimensions, <italic>f<sub>i</sub></italic> is the traces of neighboring pixels, and <italic>a<sub>i</sub></italic> is a Gaussian kernel for weighting. The relatively large values of the correlation map indicate the neuron locations. We finally fused the average images with correlation maps to obtain new images. These new images were used as inputs to the network for training and testing, corresponding to the &#x201C;Input&#x201D; for NeuroSeg-II (<xref ref-type="fig" rid="F1">Figure 1A</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>The NeuroSeg-II architecture. <bold>(A)</bold> The two-photon Ca<sup>2+</sup> image is input to the network. ResNet uses the down-sampling structure of the backbone, and we added one more down-sampling layer after the C5 layer. Feature pyramid network (FPN) uses each layer of the ResNet output feature to the input corresponding to the up-sampling structure. The channel attention mechanism module efficient channel attention (ECA) was added to lateral connections. <bold>(B)</bold> FPN+ uses the additional down-sampling structure to obtain input from FPN. The channel attention mechanism module ECA was added between lateral connections. <bold>(C)</bold> Region selection and feature aggregation subnetwork. Region proposal network (RPN) uses the FPN to extract the feature map, score the front and back scenes, and select the target region. Region of interest (ROI) Align matches the original image with the feature image for feature aggregation. <bold>(D)</bold> Head network consists of a parallel mask segmentation branch and a classification/regression branch. <bold>(E)</bold> Neuron segmentation result by NeuroSeg-II [data from allen brain observatory (ABO) dataset, experiment ID: 511510945]. <bold>(F)</bold> The convergence of the proposed network in the training and validation datasets is demonstrated over the course of 200 epochs using TensorBoard. Loss values were normalized to better visualize trends. Training dataset, orange; validation dataset, blue.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g001.tif"/>
</fig>
</sec>
<sec id="S2.SS3">
<title>2.3. Neural network architecture</title>
<p>The realization of neuron segmentation requires the accurate detection and segmentation of objects in the image. NeuroSeg-II is implemented with a combination of object detection and segmentation, with Mask R-CNN (<xref ref-type="bibr" rid="B15">He et al., 2017</xref>) being the backbone of our network model. To perform prediction for neurons in imaging data, the architecture uses ResNet (<xref ref-type="bibr" rid="B16">He et al., 2016</xref>) to extract features and the feature pyramid network (FPN) (<xref ref-type="bibr" rid="B26">Lin et al., 2017</xref>) as the feature hierarchy within the network (<xref ref-type="fig" rid="F1">Figure 1A</xref>). FPN is an important network component for detecting objects at different scales. FPN is taking the advantages of both strong semantic information from the top layers and high-resolution information from the bottom layers. This approach allows the network to have good semantic and high-resolution information at different scales and enhances the performance of object segmentation. Based on this advantage, FPN enables efficient neuron detection and segmentation at different scales.</p>
<p>As attention mechanisms have been reported to increase the power of emphasizing important objects and suppressing the background, we added an attention mechanism module to the lateral connection in the model to improve the feature extraction ability. Here, efficient channel attention (ECA) (<xref ref-type="bibr" rid="B48">Wang et al., 2019</xref>) is used to add lateral connections. The ECA consists of global average pooling (GAP) and fast one-dimensional (1D) convolutions (<xref ref-type="fig" rid="F2">Figure 2A</xref>). The GAP is used to process the obtained aggregated features, and fast 1D convolution is used to generate channel weights. ECA is a modification of squeeze-and-excitation networks (<xref ref-type="bibr" rid="B18">Hu et al., 2018</xref>), which augment appropriate cross-channel interactions and eliminate dimensionality reduction to improve channel attention.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Attention mechanism and path augmentation modules. <bold>(A)</bold> Diagram of efficient channel attention (ECA) module. GAP, Global average pooling. &#x2297;: The output is combined with the input feature map. <bold>(B)</bold> Illustration of the structure of ResNet down-sampling path augmentation. <bold>(C)</bold> Left top: &#x201C;Add&#x201D; is used as the fusion method of up-sampling and lateral connection. Left bottom: &#x201C;Concatenate&#x201D; is used as the fusion method of down-sampling and the lateral connection. Right: Illustration of the structure of the CSPLayer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g002.tif"/>
</fig>
<p>To propagate features with stronger semantics, we added a down-sampling path to shorten the information path between the top and bottom layers (<xref ref-type="bibr" rid="B27">Liu et al., 2018</xref>), and we enhanced the feature hierarchy to localize objects in the bottom layers (<xref ref-type="fig" rid="F1">Figure 1A</xref>). Briefly, based on the four times down-sampling in ResNet, we added one down-sampling after ResNet-C5 (the last layer of ResNet), which changes the original four feature outputs into five feature outputs (<xref ref-type="fig" rid="F2">Figure 2B</xref>). Following ResNet, the up-sampling and down-sampling paths also increase the input and output features, respectively. This modification reduces the loss of information and increases the utilization of feature information at diverse levels, thus enriching the feature information of small objects. To disseminate features in the network, two kinds of feature integration approaches are used (<xref ref-type="fig" rid="F2">Figure 2C</xref>), &#x201C;Add&#x201D; (<xref ref-type="bibr" rid="B15">He et al., 2017</xref>) and &#x201C;Concatenate&#x201D; (<xref ref-type="bibr" rid="B5">Bochkovskiy et al., 2020</xref>), for each layer&#x2019;s transverse connection paths. The &#x201C;Add&#x201D; increases the number of image features but does not increase the description image dimension. The &#x201C;Concatenate&#x201D; increases the features of the image, enriching the features of the image and reducing the redundancy of information (<xref ref-type="bibr" rid="B19">Huang et al., 2017</xref>). A cross-stage partial layer (CSPLayer) (<xref ref-type="bibr" rid="B46">Wang C.-Y. et al., 2020</xref>) was added after the &#x201C;Concatenate&#x201D; (<xref ref-type="fig" rid="F2">Figure 2C</xref>), which strengthens the learning ability of the network. It also eliminates the computing bottleneck and reduces the memory cost of using &#x201C;Concatenate&#x201D; multiple times (<xref ref-type="bibr" rid="B19">Huang et al., 2017</xref>).</p>
<p>To further strengthen the object localization capability for the feature hierarchy, and combine the response of higher-level neurons to the whole of objects and the response of lower-level neurons to local textures, we added a new information path from the top to bottom layers connected to FPN, which we call &#x201C;FPN+&#x201D; (<xref ref-type="fig" rid="F1">Figure 1B</xref>). The &#x201C;FPN+&#x201D; performs step-by-step down-sampling to generate new feature maps and obtain higher-resolution feature maps and coarser maps through lateral connections. Hence, the &#x201C;FPN+&#x201D; generates the new feature maps from Y<sub>2</sub> to Y<sub>6</sub>. The input image was combined with the feature map to prepare for the subsequent image segmentation (<xref ref-type="fig" rid="F1">Figure 1C</xref>). The following Head network was used to classify and segment the original image (<xref ref-type="fig" rid="F1">Figure 1D</xref>) and generate the output result (<xref ref-type="fig" rid="F1">Figure 1E</xref>). The loss function of network model combines the losses of classification, regression and segmentation mask. The region selection, feature aggregation, and Head network are the same as those in Mask R-CNN.</p>
</sec>
<sec id="S2.SS4">
<title>2.4. Data augmentation</title>
<p>To enhance the training effect, we used the imgaug tool to expand the training sample. In the training process, the original image was flipped horizontally (50% probability), flipped vertically (50% probability), rotated (90, 180, 270<sup>&#x00B0;</sup>), scaled (0.8&#x2013;1.5 times), and added with Gaussian noise (intensity from 0.0 to 5.0). We randomly used zero to five items, as mentioned above, in the training network.</p>
</sec>
<sec id="S2.SS5">
<title>2.5. Model training strategies</title>
<sec id="S2.SS5.SSS1">
<title>2.5.1. Model training with hybrid dataset</title>
<p>This training strategy was used to perform the model training and the performance testing of the attention mechanism module, the enhanced feature hierarchy, and the improvement from Mask R-CNN to NeuroSeg-II. The dataset from our lab and the ABO dataset were used, including 325 images (193 images from our lab and 132 images from the ABO dataset). It was divided into three sub-datasets: the training dataset (223 images), the validation dataset (51 images), and the testing dataset (51 images). Using a transfer learning approach, we used this hybrid dataset to train the ResNet based on the model parameters pre-trained on the ImageNet dataset. We trained NeuroSeg-II for 200 epochs. The training process was divided into three stages. The first stage froze all layers except the Head network and consisted of 50 epochs running at a learning rate of 1 &#x00D7; 10<sup>&#x2013;3</sup>. The second stage thawed the global network and consisted of 100 epochs running at a learning rate of 2 &#x00D7; 10<sup>&#x2013;4</sup>. The third stage reduced the learning rate to 1 &#x00D7; 10<sup>&#x2013;4</sup> and consisted of 50 epochs. Each epoch consisted of 500 steps with a batch size of two. By using the above strategy to train our network, it shows that the loss was decreased clearly at the initial learning stage (before 50 epochs), and was converged at the end of the learning process (<xref ref-type="fig" rid="F1">Figure 1F</xref>).</p>
</sec>
<sec id="S2.SS5.SSS2">
<title>2.5.2. Model training with the neurofinder dataset</title>
<p>To compare the neuron segmentation performance of NeuroSeg-II with other methods, we used the Neurofinder dataset for evaluation. All the methods were trained and tested by two-round hybrid cross validation. Through the evaluation of the methods using video data for training and testing, we trained the models, including SUNS (<xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>), STNeuroNet (<xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>), and CaImAn (<xref ref-type="bibr" rid="B11">Giovannucci et al., 2019</xref>), or ROI classifiers in Suite2p (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>). Five groups of video datasets (01.00&#x2013;04.01) were trained together to generate one model. We used the trained models to test the five corresponding video data (01.00.test&#x2013;04.01.test) and obtained the result, that is, the second round of cross validation of the training dataset and testing dataset interchange. We obtained the testing result of all 10 groups of the Neurofinder dataset. For the evaluation of the methods used image data for training and testing (NeuroSeg-II), we replaced the training dataset with 35 images (five groups of video datasets with seven images per dataset) and the testing dataset with five images (five groups of video dataset with one global image per dataset). We obtained the testing result of all 10 groups of the Neurofinder dataset. All the above methods were optimized according to their papers (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>; <xref ref-type="bibr" rid="B11">Giovannucci et al., 2019</xref>; <xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>; <xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>). We used the model provided by CITE-On (<xref ref-type="bibr" rid="B38">Sit&#x00E0; et al., 2022</xref>) for training and testing. To adapt to the CITE-On neuron detection function, we converted the annotated neuron edges into bounding boxes. The training process of NeuroSeg-II was as follows: The network model pre-trained with the hybrid dataset was trained with the Neurofinder dataset for a total of 150 epochs, including two stages. The first stage froze all layers except the Head network and consisted of 20 epochs running at a learning rate of 1 &#x00D7; 10<sup>&#x2013;3</sup>, and the second stage thawed the global network and consisted of 130 epochs running at a learning rate of 1 &#x00D7; 10<sup>&#x2013;3</sup>. Each epoch consisted of 50 steps with a batch size of two.</p>
<p>To verify the rationality of two-round hybrid cross validation, we also used a 10-round single cross-validation procedure for training and testing NeuroSeg-II. In the 10-round (one-to-one) cross-validation procedure, we used each of the 10 groups in the dataset as the training data only once. In each round of cross validation, we used seven preprocessed images (evenly divided part in video) as the training dataset and one image (whole video) as the test dataset (e.g., 01.00 training, 01.00.test test). NeuroSeg-II&#x2019;s training process was the same as that of the two-round hybrid cross validation. Each epoch consisted of 20 steps with a batch size of two.</p>
</sec>
</sec>
<sec id="S2.SS6">
<title>2.6. Evaluation metrics</title>
<p>To evaluate segmentation methods, we compared the results with the GT (<xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>). We performed the evaluation with three metrics (i.e., precision, recall, and F1-score), defined as follows:</p>
<disp-formula id="S2.E2">
<label>(2)</label>
<mml:math id="M2">
<mml:mrow>
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mtext>GT</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mfrac>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.E3">
<label>(3)</label>
<mml:math id="M3">
<mml:mrow>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mtext>detected</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mfrac>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.E4">
<label>(4)</label>
<mml:math id="M4">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>F1</mml:mi>
<mml:mo>-</mml:mo>
<mml:mi>score</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>&#x2009;</mml:mo>
<mml:mo>&#x00D7;</mml:mo>
<mml:mo>&#x2009;</mml:mo>
<mml:mtext>Precision</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>&#x2009;</mml:mo>
<mml:mo>+</mml:mo>
<mml:mo>&#x2009;</mml:mo>
<mml:mtext>Precision</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>N</italic><sub>TP</sub> is the number of true-positive neurons, <italic>N</italic><sub>GT</sub> is the number of manually labeled neurons, and <italic>N</italic><sub>detected</sub> is the number of detected neurons. The intersection-over-union (IoU) metric and the Hungarian algorithm were applied to calculate the degree of overlap between the detected neuron masks and the GT (<xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>). The IoU was measured with two binary masks, <italic>m<sub>1</sub></italic> and <italic>m<sub>2</sub></italic>:</p>
<disp-formula id="S2.E5">
<label>(5)</label>
<mml:math id="M5">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mtext>IoU</mml:mtext>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mtext>m</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2229;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x222A;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">|</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Then, the distance (Dist) between a pair of masks is measured as</p>
<disp-formula id="S2.Ex1">
<mml:math id="M6">
<mml:mrow>
<mml:mtext>Dist</mml:mtext>
<mml:mfenced><mml:mrow>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mtext>GT</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex30">
<label>(6)</label>
<mml:math id="M7">
<mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#x2009;</mml:mo>
<mml:mfenced close="" open="{">
<mml:mrow>
<mml:mtable equalrows='true' equalcolumns='true'>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>IoU</mml:mtext>
<mml:mo stretchy='false'>(</mml:mo>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mtext>GT</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo stretchy='false'>)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>IoU</mml:mtext>
<mml:mo stretchy='false'>(</mml:mo>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mtext>GT</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo stretchy='false'>)</mml:mo>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mtext>GT</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2286;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2004;</mml:mo>
<mml:mtext>or</mml:mtext>
<mml:mo>&#x2004;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2286;</mml:mo>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mtext>GT</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x221E;</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>otherwise</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula><mml:math id="INEQ1"><mml:msubsup><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mtext>GT</mml:mtext></mml:mrow></mml:msubsup></mml:math></inline-formula> is the mask <italic>i</italic> for the GT, and <italic>M<sub>j</sub></italic> is mask <italic>j</italic> for the detected neuron. After that, the Hungarian algorithm was used to generate the true-positive neuron masks.</p>
</sec>
<sec id="S2.SS7">
<title>2.7. Statistical analysis</title>
<p>In this study, all summary data were expressed as the mean &#x00B1; SEM. For all statistical tests, a two-sided Wilcoxon signed-rank test was applied with MATLAB 2018b (MathWorks, USA) (&#x002A;<italic>P</italic> &#x003C; 0.05; <sup>&#x002A;&#x002A;</sup><italic>P</italic> &#x003C; 0.01; <sup>&#x002A;&#x002A;&#x002A;</sup><italic>P</italic> &#x003C; 0.001; and ns, not significant). The results were deemed statistically significant when <italic>P</italic> &#x003C; 0.05. Statistical parameters, including the definitions and exact values of n, were reported in the text and figure legends. No data was considered an outlier and removed from statistical analyses.</p>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>3. Results</title>
<p>The data preprocessing, network model training, and testing were conducted using Ubuntu 20.04.4 LTS, Intel Xeon Gold 6152 CPU, 256 GB RAM, NVIDIA Tesla V100 GPU.</p>
<sec id="S3.SS1">
<title>3.1. Image fusion to represent neuronal spatiotemporal activity</title>
<p>The 2D image generated from functional imaging data can represent the temporal information in the whole video by average projection (<xref ref-type="bibr" rid="B43">Stoyanov et al., 2018</xref>), maximum projection (<xref ref-type="bibr" rid="B36">Shen et al., 2018</xref>), or correlation map (<xref ref-type="bibr" rid="B42">Spaen et al., 2019</xref>; <xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>). In the average image, the neurons often have a &#x201C;donut&#x201D; structure (<xref ref-type="bibr" rid="B30">Pachitariu et al., 2013</xref>; <xref ref-type="bibr" rid="B3">Apthorpe et al., 2016</xref>; <xref ref-type="fig" rid="F3">Figure 3A</xref>) because Ca<sup>2+</sup> indicators are normally expressed in the cytoplasm of a neuron (<xref ref-type="bibr" rid="B7">Chen et al., 2013</xref>). However, this criterion is insufficient because sparsely firing neurons are invisible (<xref ref-type="fig" rid="F3">Figure 3B</xref>) in the average image (<xref ref-type="bibr" rid="B44">Stringer and Pachitariu, 2019</xref>). In contrast, they are visible in the correlation map of each pixel with its near pixels (<xref ref-type="bibr" rid="B39">Smith and Hausser, 2010</xref>; <xref ref-type="bibr" rid="B35">Portugues et al., 2014</xref>; <xref ref-type="fig" rid="F3">Figures 3A, B</xref>). Moreover, the opposite is also true: Many neurons visible in the average image are invisible in the correlation map, suggesting that their fluorescence only reflects the baseline Ca<sup>2+</sup> in these neurons (<xref ref-type="bibr" rid="B44">Stringer and Pachitariu, 2019</xref>). Hence, the previous neuron segmentation methods with 2D images had unsatisfactory recognition accuracy for overlapping and sparsely firing neurons (<xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>). We propose image fusion as the preprocessing method to tackle this problem. We used image fusion as the preprocessing method to enhance the representation power in 2D image data. The fusion of the two kinds of images enriched the neuron spatial features in 2D images and recovered some lost temporal information in the average image (<xref ref-type="fig" rid="F3">Figures 3A, C</xref>). Therefore, we used the preprocessed images as the training and testing dataset for NeuroSeg-II.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Image fusion to improve spatiotemporal information representation. <bold>(A)</bold> The images are the average image and correlation map over the recording of one dataset, 02.00 test, from Neurofinder data. Scale bars, 50 &#x03BC;m. <bold>(B)</bold> Some invisible, sparsely firing neurons are in the average image, and the correlation map can make these neurons visible. The GT is from STNeuroNet. The yellow outlines indicate the GT neurons. Scale bars, 20 &#x03BC;m. <bold>(C)</bold> The average image and the correlation map were fused and supplemented with the spatiotemporal information of the 2D image. The GT is from STNeuroNet. The yellow outlines indicate the GT neurons. Scale bars, 50 &#x03BC;m.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g003.tif"/>
</fig>
</sec>
<sec id="S3.SS2">
<title>3.2. The attention mechanism and modified feature hierarchy improved the neuron segmentation performance</title>
<p>For the neuronal population imaging data, the uneven background can generate some neuron-like structure and thus affect the segmentation task (<xref ref-type="fig" rid="F4">Figure 4A</xref>). To focus the neurons and exclude the background influence, we added an ECA-based attention mechanism module to the lateral connection process in our network model (<xref ref-type="fig" rid="F1">Figure 1A</xref>). Here, each channel of the module plays the role of a feature detector to focus on significant parts of the input image. To observe the regions that are important for detecting neurons, we visualized how the attention module emphasizes features (<xref ref-type="fig" rid="F4">Figure 4</xref>). The figure shows that the attention mechanism could concentrate on multiple objects instead of on a single object. We can also clearly see that the masks from ECA (<xref ref-type="fig" rid="F4">Figure 4D</xref>) covered the neuron regions better than the method without using an attention mechanism (<xref ref-type="fig" rid="F4">Figure 4B</xref>) for different numbers of neurons (<italic>n</italic> = 5, 8, 18, 30 and <italic>n</italic> &#x003E; 30) in the field of view (FOV). That is, the ECA-integrated network learns well to exploit information in neuron regions and aggregate features from them. The observations confirm that the feature refinement process of ECA eventually leads networks to use the given features well. The method without using an attention mechanism only focuses on a few neurons, leading to a significant loss of object information. In contrast, unlike channel attention, the spatial attention module screens the input image location information. Hence, we compared ECA with the convolutional block attention module (CBAM) (<xref ref-type="fig" rid="F4">Figure 4C</xref>; <xref ref-type="bibr" rid="B49">Woo et al., 2018</xref>) and the model without using an attention mechanism (<xref ref-type="fig" rid="F5">Figure 5A</xref>). The CBAM uses a combination of channel attention and spatial attention mechanisms. The comparison results show that the ECA in NeuroSeg-II achieved a significantly higher F1-score than that of the other two methods (<italic>P</italic> &#x003C; 0.001, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 51 images). Hence, augmenting spatial attention mechanism did not screen the location information of neurons well, so ECA was better than CBAM and the method without using an attention mechanism for neuron segmentation.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Visualization of attention mechanisms on different numbers of neurons. <bold>(A)</bold> The input two-photon images with different numbers of neurons. <bold>(B&#x2013;D)</bold> The difference in image feature focusing between no attention module <bold>(B)</bold>, convolutional block attention module (CBAM) <bold>(C)</bold>, and efficient channel attention (ECA) <bold>(D)</bold>. We use Grad-CAM for the visualization of attention effects (n is the number of neurons in each image).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Attention mechanism, FPN+, and path augmentation increase neuron segmentation accuracy. <bold>(A)</bold> The efficient channel attention (ECA) module within our model was superior to other approaches for neuron segmentation (&#x002A;&#x002A;<italic>P</italic> &#x003C; 0.01, &#x002A;&#x002A;&#x002A;<italic>P</italic> &#x003C; 0.001; <italic>n</italic> = 51 images; and error bars are SEM). <bold>(B)</bold> NeuroSeg-II&#x2019;s neuron segmentation score was superior to those of Mask region-based convolutional neural network (R-CNN) and Mask R-CNN FPN+ (adding FPN+ to Mask R-CNN) (&#x002A;<italic>P</italic> &#x003C; 0.05, &#x002A;&#x002A;<italic>P</italic> &#x003C; 0.01, &#x002A;&#x002A;&#x002A;<italic>P</italic> &#x003C; 0.001, and <italic>n</italic> = 51 images; ns, not significant; and error bars are SEM). The results were obtained using hybrid training. <bold>(C)</bold> The neuron segmentation score of using &#x201C;Add&#x201D; for up-sampling and &#x201C;Concatenate&#x201D; for down-sampling was superior to those of other methods (&#x002A;<italic>P</italic> &#x003C; 0.05, &#x002A;&#x002A;<italic>P</italic> &#x003C; 0.01, and &#x002A;&#x002A;&#x002A;<italic>P</italic> &#x003C; 0.001; <italic>n</italic> = 51 images; ns, not significant; and error bars are SEM). The results were obtained using hybrid training. <bold>(D)</bold> The ablation study for testing the network components including attention module, path augmentation and &#x201C;FPN+&#x201D; (&#x002A;&#x002A;<italic>P</italic> &#x003C; 0.01; <italic>n</italic> = 10 images; ns, not significant; and error bars are SEM). The results were obtained using two-round hybrid cross validation with the Neurofinder dataset. All <italic>P</italic>-values were calculated with a two-sided Wilcoxon signed-rank test. The gray dots represent the scores for each testing image.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g005.tif"/>
</fig>
<p>In two-photon imaging experiments, the recorded neurons could be small and dense in a large FOV, so neurons contain too few discriminative features owing to the few pixels (<xref ref-type="bibr" rid="B23">Kisantal et al., 2019</xref>). A small object is defined as an object with pixel values of less than 32 &#x00D7; 32 (<xref ref-type="bibr" rid="B6">Bosquet et al., 2018</xref>), and the diameter of neurons from ABO and Neurofinder datasets being 10&#x2013;30 pixel values. A previous study (<xref ref-type="bibr" rid="B50">Zeiler and Fergus, 2014</xref>) reported that higher-level objects were activated entirely. In contrast, lower-level objects were activated locally. This indicates that it is necessary to augment the top-down information path to propagate features and enhance feature extraction capability in FPN. To enhance feature extraction at multi-scales, we added &#x201C;FPN+&#x201D; and path augmentation to the feature hierarchy in the network model (<xref ref-type="fig" rid="F1">Figures 1A, B</xref>).</p>
<p>To further validate the path augmentation effect, we compared NeuroSeg-II with Mask R-CNN and Mask R-CNN FPN+ (a modified version with adding &#x201C;FPN+&#x201D; to Mask R-CNN) for neuron segmentation (<xref ref-type="fig" rid="F5">Figure 5B</xref>). The results demonstrate that the precision and F1-score of NeuroSeg-II were significantly higher than those of the other two methods (<italic>P</italic> &#x003C; 0.05, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 51 images). The recall rate of NeuroSeg-II (0.917 &#x00B1; 0.014) was slightly lower than that of Mask R-CNN FPN+ (0.928 &#x00B1; 0.014; <italic>P</italic> = 0.0665, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 51 images) and significantly higher than that of Mask R-CNN (<italic>P</italic> &#x003C; 0.001, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 51 images).</p>
<p>Our network mode uses &#x201C;Add&#x201D; as the up-sampling and &#x201C;Concatenate&#x201D; as the down-sampling fusion approach (<xref ref-type="fig" rid="F2">Figure 2C</xref>). To validate the performance of this integration approach, we compared it with the other three combination options (<xref ref-type="fig" rid="F5">Figure 5C</xref>). The results show that the combination of &#x201C;Add&#x201D; as the up-sampling method and &#x201C;Concatenate&#x201D; as the down-sampling method had the highest performance to carry out feature integration. This method&#x2019;s precision was significantly higher than that of the other three approaches (<italic>P</italic> &#x003C; 0.05, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 51 images). The recall rate (0.917 &#x00B1; 0.014) was lower than those of the other three methods (0.924 &#x00B1; 0.016, 0.926 &#x00B1; 0.014, and 0.922 &#x00B1; 0.013; <italic>P</italic> &#x003C; 0.05, <italic>P</italic> = 0.1449, and <italic>P</italic> = 0.6879; two-sided Wilcoxon signed-rank test, <italic>n</italic> = 51 images). The F1-score (0.882 &#x00B1; 0.010) was significantly higher than &#x201C;All-Add&#x201D; and &#x201C;All-Concatenate&#x201D; (0.863 &#x00B1; 0.013 and 0.870 &#x00B1; 0.011; <italic>P</italic> &#x003C; 0.05, two-sided Wilcoxon signed rank test, <italic>n</italic> = 51 images), and higher than &#x201C;Up-Concatenate/Down-Add&#x201D; (0.872 &#x00B1; 0.010; <italic>P</italic> = 0.1190, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 51 images). The improvement of the effect of this combination method is due to richer image features during up-sampling that was used to increase the number of features by &#x201C;Add.&#x201D; Thus, the higher semantic information of features during down-sampling was used to enrich features by &#x201C;Concatenate.&#x201D;</p>
<p>In addition, we conducted ablation study to examine the effects of the attention mechanism and the modified feature hierarchy including path augmentation and &#x201C;FPN+&#x201D; (<xref ref-type="fig" rid="F5">Figure 5D</xref>). The results show that all the network components affected precision and recall rate. In particular, the addition of FPN+ had an obvious effect on the recall rate. The precision of NeuroSeg-II (0.731 &#x00B1; 0.029) was at the same level as that of ablating FPN+ (0.738 &#x00B1; 0.030; <italic>P</italic> = 0.9219, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images) and was higher than other ablation results (attention ablation: 0.708 &#x00B1; 0.033, <italic>P</italic> = 0.2754; path augmentation ablation: 0.710 &#x00B1; 0.030, <italic>P</italic> = 0.0781; two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images). The recall rate (0.591 &#x00B1; 0.036) and F1-score (0.644 &#x00B1; 0.022) of NeuroSeg-II were significantly higher than those of ablating FPN+ (<italic>P</italic> = 0.002 for both recall rate and F1-score, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images) and higher than other ablation results (recall rate: attention ablation, 0.586 &#x00B1; 0.044, <italic>P</italic> = 0.6523; path augmentation ablation, 0.579 &#x00B1; 0.032, <italic>P</italic> = 0.4258; F1-score: attention ablation, 0.624 &#x00B1; 0.025, <italic>P</italic> = 0.1055; path augmentation ablation, 0.627 &#x00B1; 0.017, <italic>P</italic> = 0.1055; two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images). Therefore, the results indicate that the ablation of network components resulted in lower accuracy, all network components contributed to efficient neuron segmentation.</p>
</sec>
<sec id="S3.SS3">
<title>3.3. NeuroSeg-II achieved accurate and generalized neuron segmentation</title>
<p>Based on the attention mechanism and enhanced feature hierarchy, we trained our NeuroSeg-II with a hybrid dataset and tested it for a generalized neuron segmentation task. As the training dataset contains image features from different Ca<sup>2+</sup> indicators, imaging scales, neuron activation, brain regions, imaging depths, and labs, NeuroSeg-II performed comprehensive learning and achieved generalized neuron segmentation ability. After hybrid training, we tested two-photon imaging datasets, including three Ca<sup>2+</sup> indicators (OGB-1, Cal-520, and GCaMP6) and various imaging scales. The results show that the NeuroSeg-II model achieved good performance across three Ca<sup>2+</sup> indicators and imaging scales (<xref ref-type="fig" rid="F6">Figure 6A</xref>) (Cal-520: F1-score = 0.9524; OGB-1: F1-score = 0.8536; and GCaMP6f: F1-score = 0.9171). In addition, we investigated the activities of the segmented neurons. The results (<xref ref-type="fig" rid="F6">Figures 6B&#x2013;D</xref>) exhibit that both the active and inactive neurons were recognized by our model, which suggests that the image fusion preprocessing integrates spatiotemporal information and contributes to the improvement of segmentation performance.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Results of NeuroSeg-II performing generalized neuron segmentation. <bold>(A)</bold> Examples showing the neuron segmentation results for Cal-520 (precision = 0.9677, recall = 0.9375, and F1-score = 0.9524; the dataset from our lab), OGB-1 (precision = 0.875, recall = 0.8333, and F1-score = 0.8536; the dataset from our lab), and GCaMP6f (precision = 0.9779, recall = 0.8634, and F1-score = 0.9207; the dataset from ABO dataset, experiment ID: 536323956). The yellow outlines indicate the GT neurons, and the red outlines indicate the neurons detected by NeuroSeg-II. Cal-520: scale bar, 20 &#x03BC;m. OGB-1: scale bar, 50 &#x03BC;m; GCaMP6f: scale bar, 50 &#x03BC;m. <bold>(B&#x2013;D)</bold> Examples showing the neuron segmentation of active and inactive neurons in the imaging data for <bold>(B)</bold> Cal-520 (precision = 1.0, recall = 1.0, and F1-score = 1.0; data from our lab), <bold>(C)</bold> GCaMP6f (precision = 0.9779, recall = 0.8634, and F1-score = 0.9171; data from ABO dataset, experiment ID: 536323956), and <bold>(D)</bold> GCaMP6s (precision = 0.6490, recall = 0.7967, and F1-score = 0.7153; data from Neurofinder dataset, experiment ID: 01.01test). The left side is the segmentation result, and the right side is the activities of three representative neurons (scale bar, 10 &#x03BC;m). The yellow outlines indicate the GT neurons, and the red outlines indicate the neurons segmented by NeuroSeg-II. Cal-520: scale bar, 20 &#x03BC;m. GCaMP6f: scale bar, 50 &#x03BC;m. GCaMP6s: scale bar, 50 &#x03BC;m.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g006.tif"/>
</fig>
<p>Furthermore, we tested the neuron segmentation capability of the learned NeuroSeg-II model with the large-field mesoscopic two-photon imaging data to demonstrate the generalizability of our trained model, the enhancement of the segmentation effect for small objects, and the dataset advantage of the 2D image processing approach. As the imaging data are too large to deal with the current methods, we first split the data into 306 small images and then recombined the segmentation results from small images to reconstruct the entire FOV. We obtained the final performance by comparing the segmented results with the GT (<xref ref-type="fig" rid="F7">Figure 7A</xref>). We achieved an F1-score of 0.80 (precision: 0.84; recall: 0.76). The result shows that the NeuroSeg-II model obtained by hybrid training could deal with the large-field imaging data with small neurons (<xref ref-type="fig" rid="F7">Figure 7B</xref>), which is challenging for a neuron segmentation task. Moreover, the result also demonstrates that our model maintained reliable performance for untrained images. Therefore, the learned NeuroSeg-II model integrates various neuron characteristics in imaging data and lays a foundation for generalized cell recognition and segmentation.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Neuron segmentation in mesoscopic two-photon Ca<sup>2+</sup> imaging with NeuroSeg-II. <bold>(A)</bold> Segmentation results overlaid on the imaging data of GCaMP6s expressing neurons [mesoscopic data from the study of <xref ref-type="bibr" rid="B40">Sofroniew et al. (2016)</xref>]. The yellow outlines indicate the ground truth (GT) neurons, and the red outlines indicate the neurons segmented by NeuroSeg-II. Two regions are highlighted by the white squares and shown at an expanded spatial scale. Scale bar, 1 mm. <bold>(B)</bold> The effect of neuron segmentation at different scales. B1 scale bar, 20 &#x03BC;m. B2 scale bar, 50 &#x03BC;m.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g007.tif"/>
</fig>
</sec>
<sec id="S3.SS4">
<title>3.4. Comparison of NeuroSeg-II with other neuron segmentation methods</title>
<p>To further validate our proposed network model, we compared NeuroSeg-II with other methods by testing a publicly available two-photon Ca<sup>2+</sup> imaging dataset (the Neurofinder Challenge dataset). We converted the video data from the Neurofinder dataset into images and used the datasets for training and testing the different methods. Owing to the difference between the image and video for training and testing, we conducted preprocessing and data augmentation on Neurofinder image data to compensate for the information from the videos.</p>
<p>Here, we used two-round hybrid cross validation for comparison with other methods. For all other methods, we optimized them using the algorithmic parameters mentioned in the relevant literatures (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>; <xref ref-type="bibr" rid="B11">Giovannucci et al., 2019</xref>; <xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>; <xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Sit&#x00E0; et al., 2022</xref>). The representative image (<xref ref-type="fig" rid="F8">Figure 8A</xref>) with segmented neurons and the GT demonstrate that our network achieved promising performance for this challenging dataset. Based on the 10 videos in the dataset (<xref ref-type="fig" rid="F8">Figure 8B</xref>), NeuroSeg-II achieved higher but statistically insignificant precision (0.731 &#x00B1; 0.029) than SUNS (0.633 &#x00B1; 0.067) and STNeuroNet (0.589 &#x00B1; 0.043), and significantly higher than Suite2p (0.548 &#x00B1; 0.046) CaImAn (0.539 &#x00B1; 0.045) and CITE-On (0.519 &#x00B1; 0.022) (<italic>P</italic> &#x003C; 0.01, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images). NeuroSeg-II&#x2019;s recall rate (0.591 &#x00B1; 0.036) was lower than those of the other methods (SUNS: 0.660 &#x00B1; 0.031; STNeuroNet: 0.669 &#x00B1; 0.049; Suite2p: 0.598 &#x00B1; 0.042; CaImAn: 0.5970 &#x00B1; 0.049; CITE-On: 0.724 &#x00B1; 0.030, <italic>P</italic> = 0.0195, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images). NeuroSeg-II&#x2019;s F1-score (0.644 &#x00B1; 0.022) was higher, but statistically insignificantly, than those of SUNS (0.619 &#x00B1; 0.039, <italic>P</italic> = 0.4922), STNeuroNet (0.598 &#x00B1; 0.012, <italic>P</italic> = 0.1602), Suite2p (0.552 &#x00B1; 0.031, <italic>P</italic> = 0.0645), and CITE-On (0.602 &#x00B1; 0.022, <italic>P</italic> = 0.2324), and significantly higher than that of CaImAn (0.537 &#x00B1; 0.027) (<italic>P</italic> = 0.0059, two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images). These comparison results again demonstrate the good generalization capability of NeuroSeg-II, as the high performance was consistent for segmenting neurons in the imaging data acquired from different labs.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption><p>NeuroSeg-II outperformed other neuron segmentation methods in accuracy on the Neurofinder dataset. <bold>(A)</bold> Top: Examples from the Neurofinder dataset (video 02.00) showing the neuron segmentation results of NeuroSeg-II (precision: 0.8272; recall: 0.6054; F1-score: 0.6991), SUNS (precision: 0.7482; recall: 0.6341; F1-score: 0.6865), STNeuroNet (precision: 0.5096; recall: 0.7669; F1-score: 0.6124), Suite2p (precision: 0.7014; recall: 0.3633; F1-score: 0.4787), CaImAn (precision: 0.5301; recall: 0.6978; F1-score: 0.6025), and CITE-On (precision: 0.5191; recall: 0.8395; F1-score: 0.6415), where the segmented neurons are overlaid on the fused image data. Scale bar, 50 &#x03BC;m. The yellow outlines indicate the GT neurons, and the other colors indicate the neurons found by the methods. Bottom: Examples of segmented neurons zoomed in on the white-boxed regions. Scale bar, 5 &#x03BC;m. <bold>(B)</bold> Statistical comparison of NeuroSeg-II with other methods (&#x002A;<italic>P</italic> &#x003C; 0.05, &#x002A;&#x002A;<italic>P</italic> &#x003C; 0.01; <italic>n</italic> = 10 images or videos; ns, not significant; and error bars are SEM). <bold>(C)</bold> Statistical comparison of two-round hybrid cross validation with 10-round single cross validation by NeuroSeg-II (&#x002A;&#x002A;<italic>P</italic> &#x003C; 0.01; <italic>n</italic> = 10 images; ns, not significant; and error bars are SEM). All <italic>P</italic>-values were calculated with a two-sided Wilcoxon signed-rank test. The gray dots represent the scores for each testing image.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncel-17-1127847-g008.tif"/>
</fig>
<p>In addition, we compared the two-round hybrid cross-validation strategies with a 10-round single cross-validation strategy. The results (<xref ref-type="fig" rid="F8">Figure 8C</xref>) show that the two-round hybrid cross-validation strategy achieved higher performance than the 10-round single cross-validation strategy (precision: <italic>P</italic> = 0.0645; recall: <italic>P</italic> = 0.1309; F1-score: <italic>P</italic> = 0.002; two-sided Wilcoxon signed-rank test, <italic>n</italic> = 10 images). Although the 10-round single cross validation was targeted for the same laboratory image features and labels (<xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>; <xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>), the two-round hybrid cross validation integrated all data information to enrich the image feature for network learning and improve the neuron segmentation performance.</p>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>4. Discussion</title>
<p>Here, we presented an automated, accurate, and efficient neuron segmentation method for two-photon Ca<sup>2+</sup> imaging data. The proposed network model was developed based on Mask R-CNN and modified with an attention mechanism and feature hierarchy. We used image fusion preprocessing to integrate spatiotemporal information into 2D images. Our method accurately segments both active and inactive neurons across Ca<sup>2+</sup> indicators, imaging scales, brain regions, and imaging depths with different experimental setups. Our method was also successfully applied to a large-field mesoscopic image dataset, which is challenging for neuron segmentation. For testing with the Neurofinder dataset, our approach surpassed the performance of the state-of-the-art methods (SUNS, STNeuroNet, Suite2p, CaImAn, and CITE-On) and achieved the highest precision and F1-score.</p>
<p>As the attention mechanism has been reported to be efficient in learning what and where to refine features, we added an ECA-based attention module in our model to focus on neurons properly and thus enhance the feature extraction ability. Guided with the loss function of network, the integrated attention module efficiently helps the whole network by learning which information to emphasize or suppress for distinguishing neurons. In the visualization results (<xref ref-type="fig" rid="F4">Figure 4D</xref>), we saw how the module exactly focused on the targeted neuron regions in a two-photon image. The attention mechanism within a neural network is often used to identify a single and significant object (<xref ref-type="bibr" rid="B18">Hu et al., 2018</xref>; <xref ref-type="bibr" rid="B49">Woo et al., 2018</xref>; <xref ref-type="bibr" rid="B48">Wang et al., 2019</xref>) by emphasizing pivotal features and suppressing background. Our results show that the attention mechanism can also concentrate on multiple objects instead of a single object. The comparison results (<xref ref-type="fig" rid="F5">Figure 5A</xref>) reveal that the ECA module performed well for neuron segmentation. The comparison results indicate that the ECA module had significantly higher accuracy than the network without an attention module, and outperformed the CBAM module. This may be because ECA can achieve more gains for detecting small objects. The ablation study also suggests that using ECA-based attention module produced higher accuracy (<xref ref-type="fig" rid="F5">Figure 5D</xref>), particularly about the precision. These results confirm that the ECA-based network has good generalization ability for neuron segmentation, so the ECA-based attention module makes a significant improvement to our model performance.</p>
<p>To enhance the feature extraction capability, we also modified the model with a path augmentation strategy, including adding an additional down-sampling module and deepening the network structure, and we added &#x201C;FPN+&#x201D; path in the network model. The modifications in NeuroSeg-II improve the feature extraction effectively (<xref ref-type="fig" rid="F5">Figures 5B, C</xref>). This strategy enhances the high-level semantic information in the network and enriches various layers of image features in the network. The localization ability of the whole feature hierarchy is further enhanced by combining the activation of the whole object by the high-level network and the activation of the local texture by the low-level network. The receptive field of the feature extraction network is then expanded. Hence, these improvements also contributed to NeuroSeg-II&#x2019;s good performance. The ablation study also suggests that path augmentation and &#x201C;FPN+&#x201D; both contributed the segmentation performance enhancement (<xref ref-type="fig" rid="F5">Figure 5D</xref>), and &#x201C;FPN+&#x201D; is particularly useful to improve the recall rate. Some other strategies include (1) augmenting small objects directly to increase the feature information of those objects (<xref ref-type="bibr" rid="B23">Kisantal et al., 2019</xref>) and (2) detecting objects on multiscale images to ensure the consistency of scales in ImageNet (<xref ref-type="bibr" rid="B37">Singh et al., 2018</xref>). However, these strategies are not suitable for our task owing to the characteristics of neurons in two-photon Ca<sup>2+</sup> imaging data: (1) The neurons need not be augmented because of the large number in each FOV (the Neurofinder and the ABO datasets contain &#x223C;100&#x2013;400 neurons); (2) the targeted objects are neurons in images, so there are no different object classes as there are in the general object segmentation task.</p>
<p>The results indicate that NeuroSeg-II enables good segmentation accuracy along with a convenient training and testing process. This approach can avoid misidentification due to out-of-focus fluorescence [termed &#x201C;neuropil&#x201D; (<xref ref-type="bibr" rid="B32">Peron et al., 2015</xref>)] near neurons. Compared with spatiotemporal methods, our method is also able to use spatiotemporal activity information. The network model with attention mechanism and enhanced feature hierarchy provided accurate neuron segmentation across different datasets and had achieved higher F1-score than spatiotemporal methods, e.g., STNeuroNet. The results of the ablation study confirm that combining components of attention module, path augmentation and &#x201C;FPN+&#x201D; provided the best performance, and that is the reason why the proposed network model outperformed the state-of-the-art methods (SUNS, STNeuroNet, Suite2p, CaImAn, and CITE-On). In training by a hybrid imaging dataset, NeuroSeg-II can perform the neuron segmentation task with robustness and generalization ability. NeuroSeg-II was trained with images of various neuron characteristics simultaneously, and then it successfully segmented neurons from multiple datasets, including different Ca<sup>2+</sup> indicators, brain regions, or depths acquired by independent labs (<xref ref-type="fig" rid="F5">Figures 5</xref>&#x2013;<xref ref-type="fig" rid="F8">8</xref>). This training strategy also enables NeuroSeg-II to transfer the learned neuron features to new ones, which can quickly and conveniently meet the needs of experimental targeted neurons. The convenience of using NeuroSeg-II is also reflected in the fact that it does not need to adjust the parameters for various types of two-photon Ca<sup>2+</sup> imaging data. In contrast, the other four methods compared in this paper have specific requirements and adjustments on the parameters of the Neurofinder datasets (<xref ref-type="bibr" rid="B31">Pachitariu et al., 2017</xref>; <xref ref-type="bibr" rid="B11">Giovannucci et al., 2019</xref>; <xref ref-type="bibr" rid="B41">Soltanian-Zadeh et al., 2019</xref>; <xref ref-type="bibr" rid="B4">Bao et al., 2021</xref>). As a result, if the dataset is changed and the parameters are not adjusted, the segmentation performance will be degraded. We can make the network learn more neuronal features through the continuous accumulation of datasets and achieve the purpose of rapid training through a small amount of retraining.</p>
<p>Future work should extend the current network to increase processing speed and learning ability for attention-guided multiple sources and small-sample. To achieve accurate and high-speed neuron segmentation, improvements of network architecture (e.g., a light-weight network model) can potentially overcome the tradeoff between accuracy and running speed. It will be helpful to perform fast neuron segmentation and may facilitate large-scale imaging experiments (<xref ref-type="bibr" rid="B9">Fan et al., 2019</xref>). For two-photon Ca<sup>2+</sup> imaging data, the attention mechanism of multiple source domains can extract more image features and reduce image information loss. Small-sample learning can reduce the amount of data required for network learning, improve the training speed, and reduce the time cost of image data processing at the early stage. In addition, using machine learning methods to enhance signal-to-noise ratio of Ca<sup>2+</sup> imaging data will also reinforce the accuracy of neuron segmentation (<xref ref-type="bibr" rid="B25">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B51">Zhuang and Wu, 2022</xref>). These methods represent the future development of our work.</p>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The imaging data supporting the conclusions of this article are available from the corresponding authors upon reasonable request. The trained NeuroSeg-II model is freely available at <ext-link ext-link-type="uri" xlink:href="https://huggingface.co/XZH-James/NeuroSeg2/tree/main">https://huggingface.co/XZH-James/NeuroSeg2/tree/main</ext-link>. The code is provided at <ext-link ext-link-type="uri" xlink:href="https://github.com/XZH-James/NeuroSeg2">https://github.com/XZH-James/NeuroSeg2</ext-link>. We used three public datasets to evaluate the performance of neuron segmentation. We used the ABO dataset from <ext-link ext-link-type="uri" xlink:href="https://observatory.brain-map.org/visualcoding/search/overview">https://observatory.brain-map.org/visualcoding/search/overview</ext-link>. We used the Neurofinder dataset from <ext-link ext-link-type="uri" xlink:href="https://github.com/codeneuro/neurofinder">https://github.com/codeneuro/neurofinder</ext-link> and the corresponding labels created from <ext-link ext-link-type="uri" xlink:href="https://github.com/soltanianzadeh/STNeuroNet/tree/master/Markings/Neurofinder">https://github.com/soltanianzadeh/STNeuroNet/tree/master/Markings/Ne urofinder</ext-link>. We used the large-field mesoscopic two-photon imaging dataset from <ext-link ext-link-type="uri" xlink:href="https://github.com/sofroniewn/2pRAM-paper">https://github.com/sofroniewn/2pRAM-paper</ext-link>.</p>
</sec>
<sec id="S6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The two-photon Ca<sup>2 +</sup> imaging experiment with mice in our lab was approved by the Institutional Animal Care and Use Committee of Third Military Medical University. All experimental procedures were conducted in accordance with animal ethical guidelines of the Third Military Medical University Animal Care and Use Committee.</p>
</sec>
<sec id="S7" sec-type="author-contributions">
<title>Author contributions</title>
<p>XC and XL contributed to the design of the study. JP and MW performed the imaging experiments and acquired the data. ZX, YW, XC, and XL designed the method. ZX, YW, JG, SL, QH, and HJ processed the data sets. XC and XL wrote the manuscript with help from all the other authors. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec id="S8" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by grants from the National Natural Science Foundation of China (Nos. 32171096, 31925018, and 32127801).</p>
</sec>
<ack><p>We thank Jia Lou for technical assistance. XC was a junior fellow of the CAS Center for Excellence in Brain Science and Intelligence Technology.</p>
</ack>
<sec id="S9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="S10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Akerboom</surname> <given-names>J.</given-names></name> <name><surname>Carreras Calderon</surname> <given-names>N.</given-names></name> <name><surname>Tian</surname> <given-names>L.</given-names></name> <name><surname>Wabnig</surname> <given-names>S.</given-names></name> <name><surname>Prigge</surname> <given-names>M.</given-names></name> <name><surname>Tolo</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2013</year>). <article-title>Genetically encoded calcium indicators for multi-color neural activity imaging and combination with optogenetics.</article-title> <source><italic>Front. Mol. Neurosci.</italic></source> <volume>6</volume>:<issue>2</issue>. <pub-id pub-id-type="doi">10.3389/fnmol.2013.00002</pub-id> <pub-id pub-id-type="pmid">23459413</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alba</surname> <given-names>A.</given-names></name> <name><surname>Vigueras-Gomez</surname> <given-names>J. F.</given-names></name> <name><surname>Arce-Santana</surname> <given-names>E. R.</given-names></name> <name><surname>Aguilar-Ponce</surname> <given-names>R. M.</given-names></name></person-group> (<year>2015</year>). <article-title>Phase correlation with sub-pixel accuracy: A comparative study in 1D and 2D.</article-title> <source><italic>Comput. Vis. Image Und.</italic></source> <volume>137</volume> <fpage>76</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.cviu.2015.03.011</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apthorpe</surname> <given-names>N. J.</given-names></name> <name><surname>Riordan</surname> <given-names>A. J.</given-names></name> <name><surname>Aguilar</surname> <given-names>R. E.</given-names></name> <name><surname>Homann</surname> <given-names>J.</given-names></name> <name><surname>Gu</surname> <given-names>Y.</given-names></name> <name><surname>Tank</surname> <given-names>D. W.</given-names></name><etal/></person-group> (<year>2016</year>). &#x201C;<article-title>Automatic neuron detection in calcium imaging data using convolutional networks</article-title>,&#x201D; in <source><italic>Proceedings of the advances in neural information processing systems</italic></source>, (<publisher-loc>La Jolla, CA</publisher-loc>: <publisher-name>Neural Information Processing Systems</publisher-name>), <fpage>3270</fpage>&#x2013;<lpage>3278</lpage>.</citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bao</surname> <given-names>Y.</given-names></name> <name><surname>Soltanian-Zadeh</surname> <given-names>S.</given-names></name> <name><surname>Farsiu</surname> <given-names>S.</given-names></name> <name><surname>Gong</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Segmentation of neurons from fluorescence calcium recordings beyond real-time.</article-title> <source><italic>Nat. Mach. Intell.</italic></source> <volume>3</volume> <fpage>590</fpage>&#x2013;<lpage>600</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-021-00342-x</pub-id> <pub-id pub-id-type="pmid">34485824</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bochkovskiy</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>C.-Y.</given-names></name> <name><surname>Liao</surname> <given-names>H. Y. M.</given-names></name></person-group> (<year>2020</year>). <article-title>Yolov4: Optimal speed and accuracy of object detection.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. <pub-id pub-id-type="doi">10.48550/arXiv.2004.10934</pub-id> <pub-id pub-id-type="pmid">35895330</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bosquet</surname> <given-names>B.</given-names></name> <name><surname>Mucientes</surname> <given-names>M.</given-names></name> <name><surname>Brea</surname> <given-names>V. M.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>STDnet: A convnet for small target detection</article-title>,&#x201D; in <source><italic>Proceedings of the BMVC</italic></source>, <publisher-loc>Newcastle</publisher-loc>, <lpage>253</lpage>.</citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T. W.</given-names></name> <name><surname>Wardill</surname> <given-names>T. J.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Pulver</surname> <given-names>S. R.</given-names></name> <name><surname>Renninger</surname> <given-names>S. L.</given-names></name> <name><surname>Baohan</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2013</year>). <article-title>Ultrasensitive fluorescent proteins for imaging neuronal activity.</article-title> <source><italic>Nature</italic></source> <volume>499</volume> <fpage>295</fpage>&#x2013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1038/nature12354</pub-id> <pub-id pub-id-type="pmid">23868258</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dana</surname> <given-names>H.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Mohar</surname> <given-names>B.</given-names></name> <name><surname>Hulse</surname> <given-names>B. K.</given-names></name> <name><surname>Kerlin</surname> <given-names>A. M.</given-names></name> <name><surname>Hasseman</surname> <given-names>J. P.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>High-performance calcium sensors for imaging activity in neuronal populations and microcompartments.</article-title> <source><italic>Nat. Methods</italic></source> <volume>16</volume> <fpage>649</fpage>&#x2013;<lpage>657</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-019-0435-6</pub-id> <pub-id pub-id-type="pmid">31209382</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>J.</given-names></name> <name><surname>Suo</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Xie</surname> <given-names>H.</given-names></name> <name><surname>Shen</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Video-rate imaging of biological dynamics at centimetre scale and micrometre resolution.</article-title> <source><italic>Nat. Photonics</italic></source> <volume>13</volume> <fpage>809</fpage>&#x2013;<lpage>816</lpage>. <pub-id pub-id-type="doi">10.1038/s41566-019-0474-7</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Foroosh</surname> <given-names>H.</given-names></name> <name><surname>Zerubia</surname> <given-names>J. B.</given-names></name> <name><surname>Berthod</surname> <given-names>M.</given-names></name></person-group> (<year>2002</year>). <article-title>Extension of phase correlation to subpixel registration.</article-title> <source><italic>IEEE Trans. Image Process.</italic></source> <volume>11</volume> <fpage>188</fpage>&#x2013;<lpage>200</lpage>. <pub-id pub-id-type="doi">10.1109/83.988953</pub-id> <pub-id pub-id-type="pmid">18244623</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Giovannucci</surname> <given-names>A.</given-names></name> <name><surname>Friedrich</surname> <given-names>J.</given-names></name> <name><surname>Gunn</surname> <given-names>P.</given-names></name> <name><surname>Kalfon</surname> <given-names>J.</given-names></name> <name><surname>Brown</surname> <given-names>B. L.</given-names></name> <name><surname>Koay</surname> <given-names>S. A.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>CaImAn an open source tool for scalable calcium imaging data analysis.</article-title> <source><italic>Elife</italic></source> <volume>8</volume>:<issue>e38173</issue>. <pub-id pub-id-type="doi">10.7554/eLife.38173</pub-id> <pub-id pub-id-type="pmid">30652683</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grewe</surname> <given-names>B. F.</given-names></name> <name><surname>Langer</surname> <given-names>D.</given-names></name> <name><surname>Kasper</surname> <given-names>H.</given-names></name> <name><surname>Kampa</surname> <given-names>B. M.</given-names></name> <name><surname>Helmchen</surname> <given-names>F.</given-names></name></person-group> (<year>2010</year>). <article-title>High-speed in vivo calcium imaging reveals neuronal network activity with near-millisecond precision.</article-title> <source><italic>Nat. Methods</italic></source> <volume>7</volume> <fpage>399</fpage>&#x2013;<lpage>405</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1453</pub-id> <pub-id pub-id-type="pmid">20400966</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guan</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Liang</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Shi</surname> <given-names>X.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>NeuroSeg: Automated cell detection and segmentation for in vivo two-photon Ca<sup>2+</sup> imaging data.</article-title> <source><italic>Brain Struct. Funct.</italic></source> <volume>223</volume> <fpage>519</fpage>&#x2013;<lpage>533</lpage>. <pub-id pub-id-type="doi">10.1007/s00429-017-1545-5</pub-id> <pub-id pub-id-type="pmid">29124351</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harris</surname> <given-names>K. D.</given-names></name> <name><surname>Quiroga</surname> <given-names>R. Q.</given-names></name> <name><surname>Freeman</surname> <given-names>J.</given-names></name> <name><surname>Smith</surname> <given-names>S. L.</given-names></name></person-group> (<year>2016</year>). <article-title>Improving data quality in neuronal population recordings.</article-title> <source><italic>Nat. Neurosci.</italic></source> <volume>19</volume> <fpage>1165</fpage>&#x2013;<lpage>1174</lpage>. <pub-id pub-id-type="doi">10.1038/nn.4365</pub-id> <pub-id pub-id-type="pmid">27571195</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Gkioxari</surname> <given-names>G.</given-names></name> <name><surname>Doll&#x00E1;r</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Mask R-CNN</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE international conference on computer vision</italic></source>, (<publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2980</fpage>&#x2013;<lpage>2988</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.322</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Deep residual learning for image recognition</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE conference on computer vision and pattern recognition</italic></source>, (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>770</fpage>&#x2013;<lpage>778</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Helmchen</surname> <given-names>F.</given-names></name> <name><surname>Denk</surname> <given-names>W.</given-names></name></person-group> (<year>2005</year>). <article-title>Deep tissue two-photon microscopy.</article-title> <source><italic>Nat. Methods</italic></source> <volume>2</volume> <fpage>932</fpage>&#x2013;<lpage>940</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth818</pub-id> <pub-id pub-id-type="pmid">16299478</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Squeeze-and-excitation networks</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE conference on computer vision and pattern recognition</italic></source>, (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>7132</fpage>&#x2013;<lpage>7141</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00745</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>G.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Van Der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Densely connected convolutional networks</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE conference on computer vision and pattern recognition</italic></source>, (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2261</fpage>&#x2013;<lpage>2269</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.243</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>H.</given-names></name> <name><surname>Rochefort</surname> <given-names>N. L.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Konnerth</surname> <given-names>A.</given-names></name></person-group> (<year>2010</year>). <article-title>Dendritic organization of sensory input to cortical neurons in vivo.</article-title> <source><italic>Nature</italic></source> <volume>464</volume> <fpage>1307</fpage>&#x2013;<lpage>1312</lpage>. <pub-id pub-id-type="doi">10.1038/nature08947</pub-id> <pub-id pub-id-type="pmid">20428163</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>H.</given-names></name> <name><surname>Varga</surname> <given-names>Z.</given-names></name> <name><surname>Sakmann</surname> <given-names>B.</given-names></name> <name><surname>Konnerth</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>Linear integration of spine Ca<sup>2+</sup> signals in layer 4 cortical neurons in vivo.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>111</volume> <fpage>9277</fpage>&#x2013;<lpage>9282</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1408525111</pub-id> <pub-id pub-id-type="pmid">24927564</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>T. H.</given-names></name> <name><surname>Schnitzer</surname> <given-names>M. J.</given-names></name></person-group> (<year>2022</year>). <article-title>Fluorescence imaging of large-scale neural ensemble dynamics.</article-title> <source><italic>Cell</italic></source> <volume>185</volume> <fpage>9</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2021.12.007</pub-id> <pub-id pub-id-type="pmid">34995519</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kisantal</surname> <given-names>M.</given-names></name> <name><surname>Wojna</surname> <given-names>Z.</given-names></name> <name><surname>Murawski</surname> <given-names>J.</given-names></name> <name><surname>Naruniec</surname> <given-names>J.</given-names></name> <name><surname>Cho</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>Augmentation for small object detection.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. <pub-id pub-id-type="doi">10.48550/arXiv.1902.07296</pub-id> <pub-id pub-id-type="pmid">35895330</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Liao</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Yang</surname> <given-names>N.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Primary auditory cortex is required for anticipatory motor response.</article-title> <source><italic>Cereb. Cortex</italic></source> <volume>27</volume> <fpage>3254</fpage>&#x2013;<lpage>3271</lpage>. <pub-id pub-id-type="doi">10.1093/cercor/bhx079</pub-id> <pub-id pub-id-type="pmid">28379350</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>G.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>X.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Reinforcing neuron extraction and spike inference in calcium imaging using deep self-supervised denoising.</article-title> <source><italic>Nat. Methods</italic></source> <volume>18</volume> <fpage>1395</fpage>&#x2013;<lpage>1400</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-021-01225-0</pub-id> <pub-id pub-id-type="pmid">34400836</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Doll&#x00E1;r</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Hariharan</surname> <given-names>B.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Feature pyramid networks for object detection</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE conference on computer vision and pattern recognition</italic></source>, (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2117</fpage>&#x2013;<lpage>2125</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.106</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Qi</surname> <given-names>L.</given-names></name> <name><surname>Qin</surname> <given-names>H.</given-names></name> <name><surname>Shi</surname> <given-names>J.</given-names></name> <name><surname>Jia</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Path aggregation network for instance segmentation</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE conference on computer vision and pattern recognition</italic></source>, (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>8759</fpage>&#x2013;<lpage>8768</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00913</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maruyama</surname> <given-names>R.</given-names></name> <name><surname>Maeda</surname> <given-names>K.</given-names></name> <name><surname>Moroda</surname> <given-names>H.</given-names></name> <name><surname>Kato</surname> <given-names>I.</given-names></name> <name><surname>Inoue</surname> <given-names>M.</given-names></name> <name><surname>Miyakawa</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Detecting cells using non-negative matrix factorization on calcium imaging data.</article-title> <source><italic>Neural Netw.</italic></source> <volume>55</volume> <fpage>11</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2014.03.007</pub-id> <pub-id pub-id-type="pmid">24705544</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mukamel</surname> <given-names>E. A.</given-names></name> <name><surname>Nimmerjahn</surname> <given-names>A.</given-names></name> <name><surname>Schnitzer</surname> <given-names>M. J.</given-names></name></person-group> (<year>2009</year>). <article-title>Automated analysis of cellular signals from large-scale calcium imaging data.</article-title> <source><italic>Neuron</italic></source> <volume>63</volume> <fpage>747</fpage>&#x2013;<lpage>760</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2009.08.009</pub-id> <pub-id pub-id-type="pmid">19778505</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pachitariu</surname> <given-names>M.</given-names></name> <name><surname>Packer</surname> <given-names>A. M.</given-names></name> <name><surname>Pettit</surname> <given-names>N.</given-names></name> <name><surname>Dalgleish</surname> <given-names>H.</given-names></name> <name><surname>Hausser</surname> <given-names>M.</given-names></name> <name><surname>Sahani</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). &#x201C;<article-title>Extracting regions of interest from biological images with convolutional sparse block coding</article-title>,&#x201D; in <source><italic>Proceedings of the 26th international conference on neural information processing systems</italic></source>, <publisher-loc>Red Hook, NY</publisher-loc>, <fpage>1745</fpage>&#x2013;<lpage>1753</lpage>.</citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pachitariu</surname> <given-names>M.</given-names></name> <name><surname>Stringer</surname> <given-names>C.</given-names></name> <name><surname>Dipoppa</surname> <given-names>M.</given-names></name> <name><surname>Schr&#x00F6;der</surname> <given-names>S.</given-names></name> <name><surname>Rossi</surname> <given-names>L. F.</given-names></name> <name><surname>Dalgleish</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Suite2p: Beyond 10,000 neurons with standard two-photon microscopy.</article-title> <source><italic>bioRxiv</italic></source> [<comment>Preprint</comment>]. <pub-id pub-id-type="doi">10.1101/061507</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peron</surname> <given-names>S. P.</given-names></name> <name><surname>Freeman</surname> <given-names>J.</given-names></name> <name><surname>Iyer</surname> <given-names>V.</given-names></name> <name><surname>Guo</surname> <given-names>C.</given-names></name> <name><surname>Svoboda</surname> <given-names>K.</given-names></name></person-group> (<year>2015</year>). <article-title>A cellular resolution map of barrel cortex activity during tactile behavior.</article-title> <source><italic>Neuron</italic></source> <volume>86</volume> <fpage>783</fpage>&#x2013;<lpage>799</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2015.03.027</pub-id> <pub-id pub-id-type="pmid">25913859</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pnevmatikakis</surname> <given-names>E. A.</given-names></name></person-group> (<year>2019</year>). <article-title>Analysis pipelines for calcium imaging data.</article-title> <source><italic>Curr. Opin. Neurobiol.</italic></source> <volume>55</volume> <fpage>15</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1016/j.conb.2018.11.004</pub-id> <pub-id pub-id-type="pmid">30529147</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pnevmatikakis</surname> <given-names>E. A.</given-names></name> <name><surname>Soudry</surname> <given-names>D.</given-names></name> <name><surname>Gao</surname> <given-names>Y.</given-names></name> <name><surname>Machado</surname> <given-names>T. A.</given-names></name> <name><surname>Merel</surname> <given-names>J.</given-names></name> <name><surname>Pfau</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Simultaneous denoising, deconvolution, and demixing of calcium imaging data.</article-title> <source><italic>Neuron</italic></source> <volume>89</volume> <fpage>285</fpage>&#x2013;<lpage>299</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2015.11.037</pub-id> <pub-id pub-id-type="pmid">26774160</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Portugues</surname> <given-names>R.</given-names></name> <name><surname>Feierstein</surname> <given-names>C. E.</given-names></name> <name><surname>Engert</surname> <given-names>F.</given-names></name> <name><surname>Orger</surname> <given-names>M. B.</given-names></name></person-group> (<year>2014</year>). <article-title>Whole-brain activity maps reveal stereotyped, distributed networks for visuomotor behavior.</article-title> <source><italic>Neuron</italic></source> <volume>81</volume> <fpage>1328</fpage>&#x2013;<lpage>1343</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2014.01.019</pub-id> <pub-id pub-id-type="pmid">24656252</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>S. P.</given-names></name> <name><surname>Tseng</surname> <given-names>H. A.</given-names></name> <name><surname>Hansen</surname> <given-names>K. R.</given-names></name> <name><surname>Wu</surname> <given-names>R.</given-names></name> <name><surname>Gritton</surname> <given-names>H. J.</given-names></name> <name><surname>Si</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Automatic Cell segmentation by adaptive thresholding (ACSAT) for large-scale calcium imaging datasets.</article-title> <source><italic>eNeuro</italic></source> <volume>5</volume> <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1523/ENEURO.0056-18.2018</pub-id> <pub-id pub-id-type="pmid">30221189</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname> <given-names>B.</given-names></name> <name><surname>Najibi</surname> <given-names>M.</given-names></name> <name><surname>Davis</surname> <given-names>L. S.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Sniper: Efficient multi-scale training</article-title>,&#x201D; in <source><italic>Proceedings of the advances in neural information processing systems</italic></source>, <publisher-loc>Montr&#x00E9;al, QC</publisher-loc>, <fpage>9333</fpage>&#x2013;<lpage>9343</lpage>.</citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sit&#x00E0;</surname> <given-names>L.</given-names></name> <name><surname>Brondi</surname> <given-names>M.</given-names></name> <name><surname>Lagomarsino de Leon Roig</surname> <given-names>P.</given-names></name> <name><surname>Curreli</surname> <given-names>S.</given-names></name> <name><surname>Panniello</surname> <given-names>M.</given-names></name> <name><surname>Vecchia</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2022</year>). <article-title>A deep-learning approach for online cell identification and trace extraction in functional two-photon calcium imaging.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>13</volume>:<issue>1529</issue>. <pub-id pub-id-type="doi">10.1038/s41467-022-29180-0</pub-id> <pub-id pub-id-type="pmid">35318335</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>S. L.</given-names></name> <name><surname>Hausser</surname> <given-names>M.</given-names></name></person-group> (<year>2010</year>). <article-title>Parallel processing of visual space by neighboring neurons in mouse visual cortex.</article-title> <source><italic>Nat. Neurosci.</italic></source> <volume>13</volume> <fpage>1144</fpage>&#x2013;<lpage>1149</lpage>. <pub-id pub-id-type="doi">10.1038/nn.2620</pub-id> <pub-id pub-id-type="pmid">20711183</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sofroniew</surname> <given-names>N. J.</given-names></name> <name><surname>Flickinger</surname> <given-names>D.</given-names></name> <name><surname>King</surname> <given-names>J.</given-names></name> <name><surname>Svoboda</surname> <given-names>K.</given-names></name></person-group> (<year>2016</year>). <article-title>A large field of view two-photon mesoscope with subcellular resolution for in vivo imaging.</article-title> <source><italic>Elife</italic></source> <volume>5</volume>:<issue>e14472</issue>. <pub-id pub-id-type="doi">10.7554/eLife.14472</pub-id> <pub-id pub-id-type="pmid">27300105</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soltanian-Zadeh</surname> <given-names>S.</given-names></name> <name><surname>Sahingur</surname> <given-names>K.</given-names></name> <name><surname>Blau</surname> <given-names>S.</given-names></name> <name><surname>Gong</surname> <given-names>Y.</given-names></name> <name><surname>Farsiu</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Fast and robust active neuron segmentation in two-photon calcium imaging using spatiotemporal deep learning.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>116</volume> <fpage>8554</fpage>&#x2013;<lpage>8563</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1812995116</pub-id> <pub-id pub-id-type="pmid">30975747</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Spaen</surname> <given-names>Q.</given-names></name> <name><surname>Asin-Acha</surname> <given-names>R.</given-names></name> <name><surname>Chettih</surname> <given-names>S. N.</given-names></name> <name><surname>Minderer</surname> <given-names>M.</given-names></name> <name><surname>Harvey</surname> <given-names>C.</given-names></name> <name><surname>Hochbaum</surname> <given-names>D. S.</given-names></name></person-group> (<year>2019</year>). <article-title>HNCcorr: A novel combinatorial approach for cell identification in calcium-imaging movies.</article-title> <source><italic>eNeuro</italic></source> <volume>6</volume> <fpage>1</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1523/ENEURO.0304-18.2019</pub-id> <pub-id pub-id-type="pmid">31058211</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stoyanov</surname> <given-names>D.</given-names></name> <name><surname>Taylor</surname> <given-names>Z.</given-names></name> <name><surname>Carneiro</surname> <given-names>G.</given-names></name> <name><surname>Syeda-Mahmood</surname> <given-names>T.</given-names></name> <name><surname>Martel</surname> <given-names>A.</given-names></name> <name><surname>Maier-Hein</surname> <given-names>L.</given-names></name><etal/></person-group> (<year>2018</year>). <source><italic>Deep learning in medical image analysis and multimodal learning for clinical decision support.</italic></source> <publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>, <fpage>285</fpage>&#x2013;<lpage>293</lpage>.</citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stringer</surname> <given-names>C.</given-names></name> <name><surname>Pachitariu</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>Computational processing of neural recordings from calcium imaging data.</article-title> <source><italic>Curr. Opin. Neurobiol.</italic></source> <volume>55</volume> <fpage>22</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1016/j.conb.2018.11.005</pub-id> <pub-id pub-id-type="pmid">30530255</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stringer</surname> <given-names>C.</given-names></name> <name><surname>Pachitariu</surname> <given-names>M.</given-names></name> <name><surname>Steinmetz</surname> <given-names>N.</given-names></name> <name><surname>Reddy</surname> <given-names>C. B.</given-names></name> <name><surname>Carandini</surname> <given-names>M.</given-names></name> <name><surname>Harris</surname> <given-names>K. D.</given-names></name></person-group> (<year>2019</year>). <article-title>Spontaneous behaviors drive multidimensional, brainwide activity.</article-title> <source><italic>Science</italic></source> <volume>364</volume>:<issue>255</issue>. <pub-id pub-id-type="doi">10.1126/science.aav7893</pub-id> <pub-id pub-id-type="pmid">31000656</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>C.-Y.</given-names></name> <name><surname>Liao</surname> <given-names>H.-Y. M.</given-names></name> <name><surname>Wu</surname> <given-names>Y.-H.</given-names></name> <name><surname>Chen</surname> <given-names>P.-Y.</given-names></name> <name><surname>Hsieh</surname> <given-names>J.-W.</given-names></name> <name><surname>Yeh</surname> <given-names>I.-H.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>CSPNet: A new backbone that can enhance learning capability of CNN</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</italic></source>, <publisher-loc>Seattle, WA</publisher-loc>, <fpage>390</fpage>&#x2013;<lpage>391</lpage>.</citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Liao</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Liang</surname> <given-names>S.</given-names></name> <name><surname>Ding</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Single-neuron representation of learned complex sounds in the auditory cortex.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>11</volume>:<issue>4361</issue>. <pub-id pub-id-type="doi">10.1038/s41467-020-18142-z</pub-id> <pub-id pub-id-type="pmid">32868773</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Wu</surname> <given-names>B.</given-names></name> <name><surname>Zhu</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>P.</given-names></name> <name><surname>Zuo</surname> <given-names>W.</given-names></name> <name><surname>Hu</surname> <given-names>Q.</given-names></name></person-group> (<year>2019</year>). <article-title>ECA-Net: Efficient Channel Attention for Deep Convolutional Neural Networks.</article-title> <source><italic>arXiv</italic></source> [<comment>Preprint</comment>]. <pub-id pub-id-type="doi">10.48550/arXiv.1910.03151</pub-id> <pub-id pub-id-type="pmid">35895330</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Woo</surname> <given-names>S.</given-names></name> <name><surname>Park</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>J.-Y.</given-names></name> <name><surname>Kweon</surname> <given-names>I. S.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>CBAM: Convolutional block attention module</article-title>,&#x201D; in <source><italic>Proceedings of the European conference on computer vision</italic></source>, <publisher-loc>Munich</publisher-loc>, <fpage>3</fpage>&#x2013;<lpage>19</lpage>.</citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeiler</surname> <given-names>M. D.</given-names></name> <name><surname>Fergus</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Visualizing and understanding convolutional networks</article-title>,&#x201D; in <source><italic>Proceedings of the European conference on computer vision</italic></source>, (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>818</fpage>&#x2013;<lpage>833</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-10590-1_53</pub-id></citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhuang</surname> <given-names>P.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). &#x201C;<article-title>Reinforcing neuron extraction from calcium imaging data via depth-estimation constrained nonnegative matrix factorization</article-title>,&#x201D; in <source><italic>Proceedings of the 2022 IEEE international conference on image processing</italic></source>, (<publisher-loc>Bordeaux</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>216</fpage>&#x2013;<lpage>220</lpage>. <pub-id pub-id-type="doi">10.1109/ICIP46576.2022.9897521</pub-id></citation></ref>
</ref-list>
</back>
</article>