<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Behav. Neurosci.</journal-id>
<journal-title>Frontiers in Behavioral Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Behav. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-5153</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbeh.2022.759943</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Siamese Network-Based All-Purpose-Tracker, a Model-Free Deep Learning Tool for Animal Behavioral Tracking</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Su</surname> <given-names>Lihui</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1667946/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Wenyao</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1439745/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sheng</surname> <given-names>Kaiwen</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1445453/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Xiaofei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1453056/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Du</surname> <given-names>Kai</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1445441/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Tian</surname> <given-names>Yonghong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1450254/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Ma</surname> <given-names>Lei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1445881/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Computer Science, Peking University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Beijing Academy of Artificial Intelligence</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Institute for Artificial Intelligence, Peking University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Peng Cheng Laboratory</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Bart R. H. Geurten, University of G&#x00F6;ttingen, Germany</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Wenguan Wang, ETH Z&#x00FC;rich, Switzerland; Yuanyuan Mi, Chongqing University, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Yonghong Tian, <email>yhtian@pku.edu.cn</email></corresp>
<corresp id="c002">Lei Ma, <email>lei.ma@pku.edu.cn</email></corresp>
<fn fn-type="equal" id="fn001"><p><sup>&#x2020;</sup>These authors have contributed equally to this work and share first authorship</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Individual and Social Behaviors, a section of the journal Frontiers in Behavioral Neuroscience</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>16</volume>
<elocation-id>759943</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Su, Wang, Sheng, Liu, Du, Tian and Ma.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Su, Wang, Sheng, Liu, Du, Tian and Ma</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Accurate tracking is the basis of behavioral analysis, an important research method in neuroscience and many other fields. However, the currently available tracking methods have limitations. Traditional computer vision methods have problems in complex environments, and deep learning methods are hard to be applied universally due to the requirement of laborious annotations. To address the trade-off between accuracy and universality, we developed an easy-to-use tracking tool, Siamese Network-based All-Purpose Tracker (SNAP-Tracker), a model-free tracking software built on the Siamese network. The pretrained Siamese network offers SNAP-Tracker a remarkable feature extraction ability to keep tracking accuracy, and the model-free design makes it usable directly before laborious annotations and network refinement. SNAP-Tracker provides a &#x201C;tracking with detection&#x201D; mode to track longer videos with an additional detection module. We demonstrate the stability of SNAP-Tracker through different experimental conditions and different tracking tasks. In short, SNAP-Tracker provides a general solution to behavioral tracking without compromising accuracy. For the user&#x2019;s convenience, we have integrated the tool into a tidy graphic user interface and opened the source code for downloading and using (<ext-link ext-link-type="uri" xlink:href="https://github.com/slh0302/SNAP">https://github.com/slh0302/SNAP</ext-link>).</p>
</abstract>
<kwd-group>
<kwd>behavioral tracking</kwd>
<kwd>deep learning</kwd>
<kwd>model-free</kwd>
<kwd>universality</kwd>
<kwd>Siamese network</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="52"/>
<page-count count="12"/>
<word-count count="8558"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Living organisms receive cues from external environments, process the information internally, and finally output the processing outcomes in the form of behavior. Therefore quantitatively modeling and analyzing behavior is vital to help understand the motivations and underlying mechanisms of animals and is thus widely used in neuroscience (<xref ref-type="bibr" rid="B13">Frye and Dickinson, 2004</xref>; <xref ref-type="bibr" rid="B24">Krakauer et al., 2017</xref>) and other animal-related disciplines, such as psychology (<xref ref-type="bibr" rid="B47">Snowdon, 1983</xref>; <xref ref-type="bibr" rid="B10">Dewsbury, 1992</xref>), ecology (<xref ref-type="bibr" rid="B33">Nathan et al., 2008</xref>; <xref ref-type="bibr" rid="B6">Dall et al., 2012</xref>). The recent decades have witnessed the application of technology in recording and observing animal behavior, which has greatly liberated human labor in behavioral data acquisition, and yielded large amounts of data with unprecedented spatial and temporal resolutions (<xref ref-type="bibr" rid="B15">Gomez-Marin et al., 2014</xref>). These explosive animal behavioral data bring significant challenges to analysis. Fortunately, automated image-based processing methods offer opportunities to solve the challenges in behavioral analysis (<xref ref-type="bibr" rid="B7">Dell et al., 2014</xref>) and open up a new field called computational ethology that aims to quantify animal behavior (<xref ref-type="bibr" rid="B1">Anderson and Perona, 2014</xref>). Accurate trajectory tracking is the first and most crucial step of behavioral analysis (<xref ref-type="bibr" rid="B38">Pereira et al., 2020</xref>).</p>
<p>Recent advances in computer vision (CV) and deep learning have inspired many well-behaved tracking methods. Among different algorithms developed on traditional CV techniques, background subtraction is the earliest and most commonly used by software such as ToxTrac (<xref ref-type="bibr" rid="B42">Rodriguez et al., 2017</xref>). There are also software packages that apply other efficient object segmentation methods, such as the adaptive thresholding in Tracktor (<xref ref-type="bibr" rid="B48">Sridhar et al., 2019</xref>). To track individuals in groups, which can be disturbed by the touching and crossing among individuals, idTracker (<xref ref-type="bibr" rid="B40">P&#x00E9;rez-Escudero et al., 2014</xref>) uses regressive features of all individuals and successfully tracks multiple individuals simultaneously. The above-mentioned methods have shown successful tracking performance in particular conditions. However, they still have some limitations, of which the most critical one is that these methods work fine only in constrained environments because of the relatively simple features extracted by their segmentation methods. Deep learning, which is the most popular method in image processing (<xref ref-type="bibr" rid="B26">LeCun et al., 2015</xref>), has provided significant breakthroughs in designing video-based animal behavior tracking algorithms (<xref ref-type="bibr" rid="B31">Mathis and Mathis, 2020</xref>). Representative examples are idTracker.ai (<xref ref-type="bibr" rid="B43">Romero-Ferrero et al., 2019</xref>) for multiple individual tracking and DeepLabCut (DLC; <xref ref-type="bibr" rid="B30">Mathis et al., 2018</xref>), LEAP (<xref ref-type="bibr" rid="B37">Pereira et al., 2019</xref>), and DeepPoseKit (<xref ref-type="bibr" rid="B17">Graving et al., 2019</xref>) for high-dimensional postures tracking. The outstanding feature extraction ability of deep learning significantly improves the performance of tracking tools in complex environments. However, a common problem for both traditional and deep learning methods is their performance loss in &#x201C;open&#x201D; conditions, in which the statistical distributions of test datasets are different from those of training datasets (<xref ref-type="bibr" rid="B16">Goodfellow et al., 2014</xref>; <xref ref-type="bibr" rid="B34">Nguyen et al., 2015</xref>). The most effective solution for deep learning methods is enough training samples. Thus when applying deep learning methods in practice, researchers have to manually annotate a certain number of video frames to collect enough training samples, a well-known difficult task in biological fields requiring expertise and time. Besides, researchers should also be equipped with professional knowledge to train or fine-tune the neural networks. Therefore, solving practical problems by taking advantage of deep learning while bypassing its overdependence on data is a hot topic in the deep-learning field. We think SNAP-Tracker is a successful attempt to implement this idea in animal behavioral tracking.</p>
<p>To alleviate the burden of researchers and promote the development of behavioral analysis, in this article, we present an accurate, universal, and easy-to-use tracking software, Siamese Network based All-Purpose Tracker (SNAP-Tracker). As its name suggests, we develop SNAP-Tracker upon a pretrained Siamese network, consisting of two identical subnetworks to extract features and make comparisons (<xref ref-type="bibr" rid="B4">Bromley et al., 1993</xref>). Although developed upon deep learning methods, SNAP-Tracker works in a model-free way to track the object without premodeling it first. Thus, it no longer requires refinements after network pretraining. A region-of-interest (ROI) align and distractor learning protocol has been applied to the Siamese network to help overcome the disturbance from background information (<xref ref-type="bibr" rid="B49">Su et al., 2020</xref>). Briefly, the ROI-aligned operation can promise a smaller data loss/data gain ratio than the ordinary ROI pooling operation, so it is set before ROI pooling in the template branch to generate more accurate target features. SNAP-Tracker&#x2019;s graphic user interface (GUI) is tidy and easy to operate (<xref ref-type="supplementary-material" rid="FS1">Supplementary Figure 1</xref>). In most cases, users only need to define the tracking target with a bounding box at the beginning of the videos, just like taking a &#x201C;snapshot&#x201D; of the target, and SNAP-Tracker will use the &#x201C;snapshot&#x201D; as the beginning template to finish the following tracking procedure. Experimental results displayed that SNAP-Tracker can accomplish tracking tasks across various species and environmental conditions without compromising performance. With an additional detection module, SNAP-Tracker can behave in the &#x201C;tracking with detection&#x201D; mode, suitable for dealing with larger datasets or more complicated tracking tasks. However, different from other &#x201C;tracking by detection&#x201D; software, the detection module of SNAP-Tracker is only activated when tracking failures might happen, which can improve the overall accuracy but will not affect processing speed too much. To sum up, with SNAP-Tracker, accurate tracking, can become more accessible and more efficient.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Datasets</title>
<sec id="S2.SS1.SSS1">
<title>Mouse Freely Running Dataset</title>
<p>The dataset describes the freely running behavior of mice with their heads fixed. It consists of seven raw videos, provided by Jun Ding&#x2019;s Lab from Stanford University. The videos were captured from the side, and each one recorded 5,000 frames (896 &#x00D7; 600 pixels) for about 3 min. All experimental procedures were conducted in accordance with protocols approved by Stanford University&#x2019;s Administrative Panel on Laboratory Animal Care. We separated the seven videos into four groups according to foot illumination, roller color, and head direction (<xref ref-type="table" rid="T1">Table 1</xref>). Throughout all the seven videos, the forefoot and hindfoot on the closer side to the camera were manually annotated with bounding boxes and served as the ground truth to test the performance of the tracking tools. The dataset is available at <ext-link ext-link-type="uri" xlink:href="https://drive.google.com/file/d/1k0w_lgIBd5xIY0f63J8VfuccvHZ7spsD/view?usp=sharing">https://drive.google.com/file/d/1k0w_lgIBd5xIY0f63J8VfuccvHZ7spsD/view?usp=sharing</ext-link>.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Groups of the mouse freely running dataset.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Groups</td>
<td valign="top" align="center" colspan="3">Video characteristics<hr/></td>
<td valign="top" align="center">Examples</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">Foot illumination</td>
<td valign="top" align="center">Roller color</td>
<td valign="top" align="center">Head direction</td>
<td/>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">Dark</td>
<td valign="top" align="center">Left</td>
<td valign="top" align="center"><inline-graphic xlink:href="fnbeh-16-759943-i001.jpg"/></td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">Dark</td>
<td valign="top" align="center">Left</td>
<td valign="top" align="center"><inline-graphic xlink:href="fnbeh-16-759943-i002.jpg"/></td>
</tr>
<tr>
<td valign="top" align="left">3</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">Light</td>
<td valign="top" align="center">Left</td>
<td valign="top" align="center"><inline-graphic xlink:href="fnbeh-16-759943-i003.jpg"/></td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">Light</td>
<td valign="top" align="center">Right</td>
<td valign="top" align="center"><inline-graphic xlink:href="fnbeh-16-759943-i004.jpg"/></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="S2.SS1.SSS2">
<title>Other Datasets</title>
<p>The zebrafish dataset is a video of five freely swimming zebrafish recorded from the top. An example video of idTracker is available from <xref ref-type="bibr" rid="B39">Perez-Escudero et al. (2014)</xref>, and we downloaded it from <ext-link ext-link-type="uri" xlink:href="http://www.idtracker.es/">http://www.idtracker.es/</ext-link>. The mouse pupil dataset displays the abnormal pupil constriction behavior in the absence of intrinsic photosensitive retinal ganglion cells glutamate (<xref ref-type="bibr" rid="B22">Keenan et al., 2016</xref>). The chimpanzee dataset is a video about the tapping behavior of a chimpanzee on a keyboard we downloaded from <xref ref-type="bibr" rid="B19">Hattori et al. (2013)</xref>. The peacock spider dataset displays the courtship body behavior of peacock spiders (<xref ref-type="bibr" rid="B14">Girard et al., 2011</xref>). The blue-capped cordon-bleu dataset records the multimodal courtship of birds (<xref ref-type="bibr" rid="B36">Ota et al., 2015</xref>).</p>
</sec>
</sec>
<sec id="S2.SS2">
<title>Siamese Network-Based All-Purpose-Tracker</title>
<sec id="S2.SS2.SSS1">
<title>Overview</title>
<p>Siamese Network-based All-Purpose-Tracker was written in Python 3 and implemented with PyTorch 0.4.0. We develop the GUI with Qt 5.13.0. We have tested its availability on Ubuntu 16.04 and Windows 10. More detailed information, including the executable file, the master code, and others, can be found in the GitHub repository: <ext-link ext-link-type="uri" xlink:href="https://github.com/slh0302/SNAP">https://github.com/slh0302/SNAP</ext-link>.</p>
<p>The basic structure of SNAP-Tracker follows the framework of the Siamese network (<xref ref-type="bibr" rid="B3">Bertinetto et al., 2016</xref>), which consists of twin-deep convolution networks sharing the same set of parameters. The first essential module of SNAP-Tracker is the feature extractor pretrained on ImageNet, which can be either a 5-layer AlexNet (<xref ref-type="bibr" rid="B25">Krizhevsky et al., 2012</xref>) or a 50-layer ResNet-50 (<xref ref-type="bibr" rid="B20">He et al., 2016</xref>). Experiments in this paper were all performed with the faster AlexNet. Another essential module is the similarity metric module used for calculating the cross-correlation between the template and target frames. The area with maximum similarity will be decided as the tracking location. To decrease the disturbance of background information, we have developed an ROI align and distractor learning protocol (<xref ref-type="bibr" rid="B49">Su et al., 2020</xref>). Briefly speaking, the ROI align layer is placed after the feature extractor to maintain the template scale with more accurate features and exclude the disturbance from marginal background information. Further distractor learning is performed after cross-correlation calculation to increase the Euclidian distance between the target and distractors.</p>
<p>We have also implemented a &#x201C;tracking with detection&#x201D; mode with an additional detection module, a Faster-RCNN with a 50-layer ResNet that can be trained with the primary tracking results offered by the basic tracking module. When the detection module is activated, if the output confidence of SNAP-Tracker is below 0.3, it will help guide the tracking procedure.</p>
</sec>
<sec id="S2.SS2.SSS2">
<title>Network Training</title>
<p>We have used AlexNet (<xref ref-type="bibr" rid="B25">Krizhevsky et al., 2012</xref>) and stride-reduced ResNet50 (<xref ref-type="bibr" rid="B20">He et al., 2016</xref>) as the backbone network to perform proposal classification and bounding box regression with five anchors as in <xref ref-type="bibr" rid="B28">Li et al. (2018)</xref>. The backbone network of our architecture was pretrained on ImageNet (<xref ref-type="bibr" rid="B44">Russakovsky et al., 2015</xref>). Then we further trained the whole neural network of SNAP-Tracker on COCO (<xref ref-type="bibr" rid="B29">Lin et al., 2014</xref>), ImageNet DET (<xref ref-type="bibr" rid="B8">Deng et al., 2009</xref>), ImageNet VID (<xref ref-type="bibr" rid="B44">Russakovsky et al., 2015</xref>), and YouTube-Bounding Boxes Dataset (<xref ref-type="bibr" rid="B41">Real et al., 2017</xref>) to learn a general measurement of similarities between objects for visual tracking. In both training and testing, we used single-scale images with 127 pixels for template patches and 255 pixels for searching regions. We applied stochastic gradient descent with the momentum of 0.9 and a weight attenuation of 0.0005 as the optimizing method. We warmed up ResNet50 with a learning rate of 0.005 for the first five epochs. For AlexNet, we fixed the parameters of its first three layers, and AlexNet did not need a warm-up at the beginning of the training. Then we set 0.001 and 0.0001 as the learning rate of the backbone network and the rest of the network (<xref ref-type="bibr" rid="B52">Zhu et al., 2018</xref>; <xref ref-type="bibr" rid="B27">Li et al., 2019</xref>). The learning rate decayed exponentially to one-tenth of the original value.</p>
<p>The detection module of SNAP-Tracker is a pretrained 50-layer ResNet-based Faster-RCNN. For the retraining of the detection module, we kept all of the parameters default. We collected 10&#x2013;50K frames by the basic SNAP-Tracker for retraining, the initial learning rate was 0.001, and the batch size was 32. For DLC and LEAP, the initial learning rate was 0.005 and 0.0001, and the batch size was 16 and 8, correspondingly.</p>
</sec>
<sec id="S2.SS2.SSS3">
<title>Output Confidence</title>
<p>Output confidence represents the confidence level of the model about results. It is used in the &#x201C;tracking with detection&#x201D; mode to activate the detection module. As we can regard object tracking as a binary classification problem between the tracking target and background information, we used the classification probability of the tracking target as the output confidence.</p>
</sec>
</sec>
<sec id="S2.SS3">
<title>Experimental Design</title>
<sec id="S2.SS3.SSS1">
<title>Evaluation Criteria</title>
<p>Overlap rate (OR) is the ratio of intersection area to union area between tracking results and human annotations. We used OR to evaluate the tracking accuracy of software using bound boxes as the tracking results. We set the threshold of success at 0.5. If the OR value of the result is higher than the threshold, we can consider that the tracking is successful, and the success rate means the ratio of successful frames. We ran a test on the first 1,300 frames of mouse freely running video 1 and found out that the bounding box size in the first frame could affect the final successful rate (<xref ref-type="supplementary-material" rid="FS2">Supplementary Figure 2</xref>); and in our results, we chose circumscribed rectangle as the bounding box size by our experience.</p>
<p>We also used pixel error (PE) to evaluate the accuracy of tracking software when the tracking results of the software are points. PE is the Euclidean distance between the tracking results of software and human annotations. Positions of tracking points or bounding boxes centers refer to the tracking results of the software. We set the successful threshold at 20 pixels. If the PE value is lower than the threshold, we can consider that successful tracking and the accuracy rate means the ratio of successful frames.</p>
</sec>
<sec id="S2.SS3.SSS2">
<title>Experiments Description</title>
<p>To reveal that few human corrections are helpful to maintain high accuracy (<xref ref-type="fig" rid="F2">Figure 2</xref>), feet tracking was performed on different continuous frames (up to 3,000) of three videos from the mouse freely running dataset (<xref ref-type="table" rid="T1">Table 1</xref>). We used OR as the evaluation criterion for calculating error rates, which were the ratios of the number of failing frames to total frames (<xref ref-type="fig" rid="F2">Figure 2B</xref>). Label efforts under different tracking frames mean the ratio of human correcting frames to total frames (<xref ref-type="fig" rid="F2">Figure 2C</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>An illustration of the workflow for SNAP-Tracker. In the typical workflow of SNAP-Tracker, users should make the only annotation at the first frame of the video by dragging a bounding box out of the tracking object, which will serve as the template for the second and other later frames. In the search region of a target frame, the image feature is extracted by the pretrained feature extractor simultaneously with the template. After comparing the cross-correlation between the template and target frame feature, the similarity metric module will select the location with maximum similarity and generate an adapted bounding box outside the tracking object. Connecting the bounding boxes of all frames in series can form the object&#x2019;s trajectory. &#x002A; Denotes the similarity function (i.e., cross correlation) to be computed for target and template feature.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbeh-16-759943-g001.tif"/>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Few human corrections are helpful to increase accuracy. <bold>(A)</bold> In a practical auto-tracking procedure (the gray line in the top plot), the tracking result (the green box) can match the ground truth (the red box) most of the time (the bottom left inset). Tracking drift may occasionally happen due to the fast movement of the object or another similar object nearby (the bottom middle inset). The tracking drift can evolve to tracking failure without a correction (the bottom right inset). However, a single correction on the earliest tracking drift frame can successfully rescue the subsequent tracking failure (the black line in the top plot). <italic>X</italic>-axis: Frame number of the video; <italic>Y</italic>-axis: overlap rate with the evaluation threshold of 0.5. <bold>(B)</bold> The tracking error rates of forefoot and hindfoot increased with video frame length&#x2019;s elongation. Error rate: the percent ratio of failed frames in the total frames. <italic>N</italic> = 3 videos. <bold>(C)</bold> Fewer human corrections than error frames are enough to fix the tracking failures and achieve 100% accurate tracking results. A 100% accuracy: all the frames&#x2019; OR values are above 0.5; label effort: the percent ratio of frames needs to be corrected to keep 100% accuracy. <italic>N</italic> = 3 videos.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbeh-16-759943-g002.tif"/>
</fig>
<p>To illustrate the stability of SNAP-Tracker in open conditions, we used PE as the evaluation criterion to demonstrate the performance of SNAP-Tracker and two representative deep learning tracking tools, DLC (<xref ref-type="bibr" rid="B30">Mathis et al., 2018</xref>) and LEAP (<xref ref-type="bibr" rid="B37">Pereira et al., 2019</xref>; <xref ref-type="fig" rid="F3">Figure 3</xref>). Test frames and training frames were from the same video in the close condition test, while from different videos or under different conditions in the open condition tests. We fixed the length of the testing frames at 1,000 and repeated each test session with randomly selected clips three times.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Stable performance of SNAP-Tracker in open conditions. <bold>(A)</bold> In the close condition, training and test frames come from the same video. <bold>(B)</bold> The accuracy of three methods in the close condition with different scales of the training set. The tracking accuracy of DLC (dare blue) and LEAP (light blue) increases with more training frames. SNAP-Tracker (red) is independent of training, and its performance is comparable to the other two. Accuracy: the ratio of frames with PE value lower than the threshold of 20 pixels. <bold>(C)</bold> In four kinds of open conditions, the test video can be a different one with similar environmental conditions (the first row) or has different illumination (the second row), different roller colors (the third row), different head directions (the last row). <bold>(D)</bold> In open conditions, the accuracies of DLC and LEAP both drop significantly even trained with the highest number of training frames. However, SNAP-Tracker can still keep relatively good performance due to its independence to training.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbeh-16-759943-g003.tif"/>
</fig>
<p>To test the applicability of SNAP-Tracker in broader situations, we used five other videos coming from published data (<xref ref-type="fig" rid="F4">Figure 4</xref>). We validated and converted the tracking results to other indexes for further analysis, such as moving distance in pixel (px), moving speed in pixel per frame (px/f), area size in pixel square (px<sup>2</sup>), and angle in degree. We also recorded the times needed for human correction and exhibited it in the percentage of total frames.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Broader applicability of SNAP-Tracker. <bold>(A)</bold> Top: In the case of individual tracking among five zebrafish, the cyan box indicates the target fish, and the blue line shows its moving trajectory. Bottom: The plot shows the variation of swimming speed (in pixel per frame) of the zebrafish. <bold>(B)</bold> Top: In the case of mouse pupil tracking, the box shows the pupil of a glutamate knockout mouse. Bottom: The plot reveals the variation of pupil area (in pixel<sup>2</sup>). <bold>(C)</bold> Top: In tracking the chimpanzee finger with complex background information, it is easy to locate the finger position accurately. Bottom: Tapping behavior can be observed in the movement speed plot (in pixel per frame). <bold>(D)</bold> Top: When analyzing the courtship behavior of a peacock spider, we tracked the tips of the pair of third legs and head. Bottom: The open angle between two third legs, a sign of the &#x201C;Fan&#x201D; dance, is speculated. <bold>(E)</bold> Top: In tracking two blue-capped cordon-blues, we tracked the positions of their heads with different color boxes. Bottom: The plot represents the movement of heads with corresponding colors, which exhibits the interactive bobbing behaviors of birds. <bold>(F)</bold> The bar plot shows the ratios of human correction in tasks. Percentages of human correction (black) in the five tasks are 0.2, 0.4, 3.6, 3.8, and 1.1% respectively. The average human correction is 0.9 &#x00B1; 1.7%.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbeh-16-759943-g004.tif"/>
</fig>
<p>To compare the efficiency of the &#x201C;tracking with detection&#x201D; mode of SNAP-Tracker with other &#x201C;tracking by detection&#x201D; methods, such as DLC, we applied PE as the evaluation criterion to evaluate their success rates under different training frames (<xref ref-type="fig" rid="F5">Figure 5B</xref>). In this experiment, we used all the seven mouse running videos as a whole to test both packages. We randomly selected 60% of the seven video frames to constitute the whole training set and evenly used 2&#x2013;100% in the training sessions. For the working pipeline of DLC, training frames were precisely the handed-labeled ground truth annotations. So the label efforts equaled its training samples. For the working pipeline of the detection mode of SNAP-Tracker, the training frames were from its immediate automatic tracking results and occasional human corrections; and we took the human corrections as the label efforts it needed.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Tracking with detection mode. <bold>(A)</bold> In the tracking with detection mode, a video can be tracked by the auto-tracking mode first, as described above. Then the detection module can be trained by these annotation results and help improve the automatic tracking with higher accuracy. Iteratively, it is possible to track a long video stably with the well-trained detection module. We set the threshold at 0.3 for output confidence activating the detection module to avoid compromising processing speed. <bold>(B)</bold> The comparison of labeling efforts between the tracking with detection mode of SNAP-Tracker (red) and DLC (blue). The accuracy of DLC is positively related to the number of training images, which requires manual annotation. While for the detection module of SNAP-Tracker, the primary tracking module can provide most annotations, achieving similar accuracy with fewer human laborious by almost two magnitudes. Accuracy: the ratio of frames with PE value lower than the threshold of 20 pixels. <bold>(C)</bold> An example of autocorrection by the detection module. The tracking with detection mode can (the black line) prevent failures that happen in the tracking-only mode (the gray line). <italic>X</italic>-axis: Frame number of the video; <italic>Y</italic>-axis: overlap rate with the evaluation threshold of 0.5.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbeh-16-759943-g005.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Framework and Workflow of Siamese Network-Based All-Purpose-Tracker</title>
<p>We developed SNAP-Tracker on a deep Siamese neural network, one of the deep neural networks widely used in visual tracking. SNAP-Tracker is a model-free tracker and thus can complete tracking tasks without modeling the object priorly, different from other model-based deep learning methods. There are two critical compositions in the basic framework of SNAP-Tracker (<xref ref-type="fig" rid="F1">Figure 1</xref>). The feature extraction module (the orange part in <xref ref-type="fig" rid="F1">Figure 1</xref>) is thoroughly pretrained first and then used for feature extraction from the bounding box of the template frame and the searching areas in target frames. The similarity metric module (the blue part in <xref ref-type="fig" rid="F1">Figure 1</xref>) determines the object&#x2019;s location by calculating the cross-correlation between the extracted features from the template and the target frames. To start a realistic tracking procedure, users can label the object in the first frame as the template and then give hands to SNAP-Tracker, which will automatically label-size adaptive bounding boxes outside the tracking object according to the maximum feature similarity to the template. With the sliding of video frames, SNAP-Tracker annotates each frame continuously and finally produces the trajectory of the tracking object through the video. <xref ref-type="supplementary-material" rid="VS1">Supplementary Video 1</xref> shows a practical case of the workflow, which is easy to operate. As described above, throughout the whole tracking procedure, usually the only thing users have to do is define their interested tracking objects at the first frame with a bounding box and then handing over the task to SNAP-Tracker by simply clicking the starting button in the GUI (<xref ref-type="supplementary-material" rid="FS1">Supplementary Figure 1</xref>).</p>
</sec>
<sec id="S3.SS2">
<title>Few Human Corrections Are Helpful to Keep High Accuracy</title>
<p>Tracking failure is a common problem for tracking tools, which can happen when the tracking object moves too fast or when another similar object occurs nearby. In these cases, the software would accumulate errors without human interference. Therefore we integrated a manually auxiliary correction module into the basic operation panel of SNAP-Tracker (the dashed box E in <xref ref-type="supplementary-material" rid="FS1">Supplementary Figure 1</xref>). Users can rescue tracking failures by stopping the tracking and correcting the error with a new bounding box, which will change the original template into a new annotation of the current frame. After this, we can restart tracking from the breaking point (<xref ref-type="supplementary-material" rid="VS2">Supplementary Video 2</xref> shows a practical case). We have shown the efficiency of human correction in preventing tracking errors with the first video of the mouse dataset (<xref ref-type="fig" rid="F2">Figure 2A</xref>). In this experiment, we used OR as the evaluation criteria and set the threshold at 0.5; tracking failure meant the OR was below the threshold. The OR of the 1,145th frame dropped suddenly below the threshold of 0.5, indicating a tracking failure might happen, which was the tracking drift to the other forefoot (the bottom middle inset of <xref ref-type="fig" rid="F2">Figure 2A</xref>). Without human correction, SNAP-Tracker regarded the wrong foot as the tracking object, and tracking failure could happen (the bottom right inset of <xref ref-type="fig" rid="F2">Figure 2A</xref>). Sometimes, it was probable for SNAP-Tracker to automatically relocate the target if the correct foot appeared again in the searching region of SNAP-Tracker. However, if we could timely correct the shifted bounding box at the 1,145th frame where the tracking drift started, the continuous tracking failures could be avoided to a large degree. Intuitively, failures would increase with the length of the video being longer. To reveal that few human corrections are helpful to keep high accuracy in this situation, we performed feet tracking on the mouse dataset with different continuous frames clips (up to 3,000). As expected, error rates, indicating the ratio of failed tracking frames to total frames, increased with longer clips (<xref ref-type="fig" rid="F2">Figure 2B</xref>). But a certain number of label efforts, representing the ratio of human correcting frames, were enough to keep the OR of each frame steady above the threshold, which we identified as 100% tracking accuracy (<xref ref-type="fig" rid="F2">Figure 2C</xref>). It must be noted that the OR value and final successful/error rate can be influenced by the bounding box size, as we tested on the first 1,300 frames of the video used in <xref ref-type="fig" rid="F2">Figure 2A</xref> (<xref ref-type="supplementary-material" rid="FS2">Supplementary Figure 2</xref>). Our paper used circumscribed rectangle as the bounding box covering the target while the size was as small as possible. In short, SNAP-Tracker can complete much better tracking with acceptable times of timely corrections by humans.</p>
</sec>
<sec id="S3.SS3">
<title>Stable Performance of Siamese Network-Based All-Purpose-Tracker in Open Conditions</title>
<p>Siamese Network-based All-Purpose-Tracker is a model-free tracking package. Unlike the popular model-based methods, model-free methods do not need to learn the prior knowledge of the tracking object in advance. Therefore SNAP-Tracker does not require model retraining or parameters fine-tuning, which significantly alleviates the need for manual annotation before applied. The manual annotation needed in the first frame will not modify the model parameters but will tell SNAP-Track what the tracking target is. In this way, SNAP-Tracker can express much more stable performance in open conditions compared with other model-based tracking methods. To demonstrate the stable performance of SNAP-Tracker in open-conditions without enough training samples, we tested SNAP-Tracker and two other representative deep learning methods, DLC (<xref ref-type="bibr" rid="B30">Mathis et al., 2018</xref>) and LEAP (<xref ref-type="bibr" rid="B37">Pereira et al., 2019</xref>), on the mouse dataset (<xref ref-type="fig" rid="F3">Figure 3</xref>). We used PE as the evaluation criterion of accuracy in this experiment and set the successful threshold at 20 pixels. The accuracy was the ratio of successful frames. When tested in close conditions, where the test dataset came from the same video as the training dataset (<xref ref-type="fig" rid="F3">Figure 3A</xref>), SNAP-Tracker and the other two deep learning methods showed good performance (<xref ref-type="fig" rid="F3">Figure 3B</xref>). It is worth noting that the two model-based deep learning methods displayed increasing accuracies with the increment of training data. However, the performance of SNAP-Tracker was independent of training due to the model-free tracking strategy, and we expressed its accuracy with a horizontal red dashed line in the figure for a better comparison. When it came to open conditions, test datasets had different feature distributions from the training dataset, such as different videos under similar environments or different videos with different conditions (foot illumination, roller colors, and head directions) (<xref ref-type="fig" rid="F3">Figure 3C</xref>). It is evident in <xref ref-type="fig" rid="F3">Figure 3D</xref> that the performance of model-based deep learning methods dropped sharply due to the lack of model fine-tuning with test data; only DLC did not show bad accuracy in the first situation in which the test dataset was the most similar to the training dataset. However, SNAP-Tracker could exhibit better and stable performance in different open conditions (<xref ref-type="fig" rid="F3">Figure 3D</xref>). Furthermore, to clearly tell DLC and LEAP what to be tracked in the test video, we have added the first annotation of test videos, and the SNAP-Tracker was used for tracking, in each corresponding DLC and LEAP training session (<xref ref-type="supplementary-material" rid="FS3">Supplementary Figure 3</xref>). Compared with before (line with dots in <xref ref-type="supplementary-material" rid="FS3">Supplementary Figure 3</xref>), by training with one additional frame, the first annotation of test videos (smooth line in <xref ref-type="supplementary-material" rid="FS3">Supplementary Figure 3</xref>) could improve the accuracy but slightly. Therefore, SNAP-Tracker can be used directly with relatively stable performance, offering a choice for tasks with varying conditions.</p>
</sec>
<sec id="S3.SS4">
<title>Broader Applicability of Siamese Network-Based All-Purpose-Tracker</title>
<p>Siamese Network-based All-Purpose-Tracker can have good applicability across different behavioral tracking paradigms. To demonstrate the broader applicability of SNAP-Tracker, we applied it in five other videos with different species and tasks coming from published data and made further analyses based on the primary tracking results (<xref ref-type="fig" rid="F4">Figure 4</xref>). In this experiment, we used PE and the threshold of 20 pixels as the evaluation criterion of accuracy and defined the accuracy as the ratio of successful frames. We showed the averaged accuracy of 10 trails on each dataset (<xref ref-type="supplementary-material" rid="FS4">Supplementary Figure 4</xref>) and typical cases demonstrating the corresponding comparison with the ground truth (<xref ref-type="supplementary-material" rid="VS3">Supplementary Videos 3</xref>&#x2013;<xref ref-type="supplementary-material" rid="VS7">7</xref>). In the individual tracking task of zebrafish, SNAP-Tracker can accurately track one of a collective of 5 zebrafish (the cyan bounding box and blue trajectory in <xref ref-type="fig" rid="F4">Figure 4A</xref>), and we could obtain the swimming speed of the animal according to the tracking trajectory (<xref ref-type="fig" rid="F4">Figure 4A</xref>). Besides individual tracking, tracking particular body parts of an animal, such as the contraction and dilation of pupils, is also essential in neuroscience. In the tracking of mouse pupils (<xref ref-type="bibr" rid="B22">Keenan et al., 2016</xref>), we could quickly identify the state of the pupil <italic>via</italic> the area of the inscribed ellipse of each bounding box, which could reveal the role of glutamate by comparing the difference between wild type and glutamate knockout mouse (<xref ref-type="fig" rid="F4">Figure 4B</xref>). In another case of tracking the finger of a chimpanzee with a more complex background (<xref ref-type="bibr" rid="B19">Hattori et al., 2013</xref>), we could also get an accurate trace of the finger and infer the tapping frequency between alternative keys (<xref ref-type="fig" rid="F4">Figure 4C</xref>). More than that, SNAP-Tracker could also be used for more sophisticated behavioral analysis, for example, the courtship behavior of peacock spiders (<xref ref-type="bibr" rid="B14">Girard et al., 2011</xref>) and blue-capped cordon-bleu (<xref ref-type="bibr" rid="B36">Ota et al., 2015</xref>). By tracking the pair of third legs and the head of a peacock spider, we could speculate the open angle between two third legs, which served as a constituent of &#x201C;Fan&#x201D; dance, a representative courtship posture of peacock spiders (<xref ref-type="fig" rid="F4">Figure 4D</xref>). Similarly, by tracking the positions of the heads of two blue-capped cordon-bleu, we could extract out the interactive bobbing behavior between them from the video (<xref ref-type="fig" rid="F4">Figure 4E</xref>). We recorded the number of corrections needed to keep 100% accuracy during tasks and found that none of the human corrections in five tasks was larger than 4% (<xref ref-type="fig" rid="F4">Figure 4F</xref>). On average, human correction only occupied a tiny portion (0.9 &#x00B1; 1.7% on average). Taken together, with a reasonable number of human corrections, users can apply SNAP-Tracker widely in various tracking tasks.</p>
</sec>
<sec id="S3.SS5">
<title>Tracking With Detection Mode</title>
<p>As shown above, the need for human correction will increase with the elongation of tracking frames (<xref ref-type="fig" rid="F2">Figure 2B</xref>). A strategy that can liberate human efforts is required in longer videos with more complex conditions. Therefore we developed a &#x201C;tracking with detection&#x201D; mode by providing SNAP-Tracker with an additional detection module. In its brief framework (<xref ref-type="fig" rid="F5">Figure 5A</xref>), the automatic tracking results from the basic SNAP-Tracker serve as the training dataset for the detection module, and the detection module can help improve the accuracy of the basic auto-tracking module. After iterative training, the well-trained detection module can take the place of human correction when a tracking shift happens. Notably, the detection module only functions when the output confidence level of SNAP-Tracker is lower than the predefined threshold; users can set a higher threshold for better accuracy or a lower threshold for faster processing. In our experiment, we set the activation threshold at 0.3. A significant difference of the &#x201C;tracking with detection&#x201D; mode of SNAP-Tracker from DLC compared with a &#x201C;tracking by detection&#x201D; deep learning method is that SNAP-Tracker itself can offer tracking results as training data, saving much hand-labeling efforts. To demonstrate the efficiency of the &#x201C;tracking with detection&#x201D; mode of SNAP-Tracker, we tested the performance (with the criterion of PE) of SNAP-Tracker and another &#x201C;tracking by detection&#x201D; method with a synthetic free-running mouse video consisting of the seven videos (<xref ref-type="fig" rid="F5">Figure 5B</xref>). We found that the &#x201C;tracking with detection&#x201D; mode can perform well with few label efforts (red lines in <xref ref-type="fig" rid="F5">Figure 5B</xref>). However, the &#x201C;tracking by detection&#x201D; method (blue lines in <xref ref-type="fig" rid="F5">Figure 5B</xref>) needed two magnitudes higher label efforts to achieve comparable accuracy. It should be clarified here that label effort has a different source in each method. Specifically, the label efforts of DLC equaled its training samples, while we took the human correction numbers as the label efforts for SNAP-Tracker. The &#x201C;tracking with detection&#x201D; mode can improve tracking efficiency compared with the basic SNAP-Tracker. In a typical tracking case of mouse foot, the &#x201C;tracking with detection&#x201D; mode (the black line in <xref ref-type="fig" rid="F5">Figure 5C</xref>) could avoid tracking errors that happen under the regular mode (the gray line in <xref ref-type="fig" rid="F5">Figure 5C</xref>). To sum up, the &#x201C;tracking with detection&#x201D; mode can replace the role of human intervention to complete more complex tracking tasks, which are suitable for dealing with larger datasets.</p>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<p>This article presents a model-free tracking software, SNAP-Tracker, which shows robust performance under various conditions. The software has already been pretrained with publicly available datasets and requires no more parameter fine-tuning when used in practical tasks, which greatly reduces the burden of users. Considering the user communities with different backgrounds, we have integrated the software into a compact and easy-to-use GUI. The &#x201C;tracking with detection&#x201D; mode of SNAP-Tracker is more automated with the help of a detection module, and we can apply it in more complex conditions. In a word, SNAP-Tracker can be a practical choice in different kinds of behavioral tracking analysis. We will discuss the characteristics of SNAP-Tracker from the following aspects.</p>
<sec id="S4.SS1">
<title>Benefits and Drawbacks of Deep Learning</title>
<p>The benefits of using deep learning in behavioral analysis are apparent as those in other CV fields. Compared to traditional CV methods, deep learning methods can achieve much more accurate performance at the human level and even beyond. So deep learning is the current trend in many fields, and popular tracking software packages use deep learning. Nevertheless, we should notice problems such as high computational consumption and overfitting in deep learning methods cautiously (<xref ref-type="bibr" rid="B31">Mathis and Mathis, 2020</xref>). Researchers have made contributions to decreasing training efforts and increasing processing speed. DLC (<xref ref-type="bibr" rid="B30">Mathis et al., 2018</xref>) was built in the way of transfer learning upon DeeperCut (<xref ref-type="bibr" rid="B21">Insafutdinov et al., 2016</xref>), a previously established model. Soon after, LEAP tried to improve the processing speed by applying a network with much fewer layers at the price of accuracy (<xref ref-type="bibr" rid="B37">Pereira et al., 2019</xref>). The more recent DeepPoseKit made considerable progress in both speed and robustness by using a multiscale deep-learning model (<xref ref-type="bibr" rid="B17">Graving et al., 2019</xref>). Even so, laborious annotations and network fine-tuning are inevitably needed, which can be much severe if the tracking task contains multiple individuals (<xref ref-type="bibr" rid="B17">Graving et al., 2019</xref>). With enough training samples, deep learning methods can perform very well. However, if there are not enough training samples, the performance of deep learning methods will be affected. The situation in practical neuroscience research could be much more challenging. The annotation of biological samples is a well-known arduous task, requiring expertise and much time. For example, when observing the courtship behavior of songbirds (<xref ref-type="bibr" rid="B36">Ota et al., 2015</xref>), scientists are interested in only a tiny portion of the whole video frames. Making annotation is time-consuming, and training in this few-shot situation is challenging. Another problem we have to resolve is the performance loss in &#x201C;open&#x201D; conditions. Environmental conditions, such as illumination, in practice can change during the task, but we cannot label a training set including all possibilities. In some particular tasks, such as screening mutant mice (<xref ref-type="bibr" rid="B5">Brown et al., 2000</xref>), the animal&#x2019;s behavior is complex to be predefined. Thus, training in this situation will be a challenge. The idea of model-free tracking is a recently introduced solution to circumvent these drawbacks, which is the designing strategy of SNAP-Tracker.</p>
</sec>
<sec id="S4.SS2">
<title>Model-Based and Model-Free Tracking</title>
<p>The model-based and model-free dichotomy is familiar in the CV tracking field. Although idTracker (<xref ref-type="bibr" rid="B40">P&#x00E9;rez-Escudero et al., 2014</xref>), idTracker.ai (<xref ref-type="bibr" rid="B43">Romero-Ferrero et al., 2019</xref>) were used in individual tracking, and DLC (<xref ref-type="bibr" rid="B30">Mathis et al., 2018</xref>), LEAP (<xref ref-type="bibr" rid="B37">Pereira et al., 2019</xref>), and DeepPosekit (<xref ref-type="bibr" rid="B17">Graving et al., 2019</xref>) were designed for pose estimation, all of them and many other tracking tools in ethology belong to model-based tracking (<xref ref-type="bibr" rid="B50">Worrall et al., 1991</xref>), which require prior knowledge of the objects before tracking. For model-based tracking, targets in the frames of a video are detected first by object detection or segmentation methods and then connected along the temporal series to generate the moving trajectory. We call this pipeline &#x201C;tracking by detection&#x201D;; the strategy of tracking by detection can increase tracking accuracy, but at the cost of processing speed and generalization. Differently, model-free tracking (<xref ref-type="bibr" rid="B51">Zhang and Van Der Maaten, 2013</xref>) is independent of the target&#x2019;s prior modeling, and users can apply the method directly to broader tasks. Without premodeling, users can define the tracking target&#x2019;s template in the first frame and then let the software finish tracking to the end frame by frame. In this way, SNAP-Tracker can be a universal method suitable for various behavioral missions.</p>
</sec>
<sec id="S4.SS3">
<title>Individual Tracking and Pose Tracking</title>
<p>According to the analyzing resolution, we can classify behavioral analysis into different stages, from coarse to fine (<xref ref-type="bibr" rid="B38">Pereira et al., 2020</xref>), which can be summarized into two classes, individual tracking and pose tracking. They are the critical consideration for users to decide the options of tracking tools. In general, the spatiotemporal trajectory of single or multiple individuals is enough to answer questions (<xref ref-type="bibr" rid="B2">Berdahl et al., 2013</xref>; <xref ref-type="bibr" rid="B32">Mersch et al., 2013</xref>; <xref ref-type="bibr" rid="B45">Seibenhener and Wooten, 2015</xref>). To simplify the tracking of multiple individuals in a group, researchers usually labeled the targets with artificial markers (<xref ref-type="bibr" rid="B35">Ohayon et al., 2013</xref>; <xref ref-type="bibr" rid="B46">Shemesh et al., 2013</xref>), which might potentially affect animal behaviors (<xref ref-type="bibr" rid="B9">Dennis et al., 2008</xref>). By defining the model of each individual, idTracker (<xref ref-type="bibr" rid="B40">P&#x00E9;rez-Escudero et al., 2014</xref>) and its deep learning version idTracker.ai (<xref ref-type="bibr" rid="B43">Romero-Ferrero et al., 2019</xref>) make tracking unmarked targets possible. In more complex situations, researchers have to extract detailed pose information of the targets (<xref ref-type="bibr" rid="B23">Khan et al., 2012</xref>; <xref ref-type="bibr" rid="B18">Guo et al., 2015</xref>; <xref ref-type="bibr" rid="B36">Ota et al., 2015</xref>), which raises the difficulty of tracking. Representative methods, DLC (<xref ref-type="bibr" rid="B30">Mathis et al., 2018</xref>), LEAP (<xref ref-type="bibr" rid="B37">Pereira et al., 2019</xref>), and DeepPoseKit (<xref ref-type="bibr" rid="B17">Graving et al., 2019</xref>), display exemplary performance in pose estimation by utilizing the outstanding feature extraction ability of deep-learning network. However, the amount of tracking individuals is still limited due to the increased computation time. Moreover, it requires exhaustive annotating efforts to establish a training dataset with the growing individuals (<xref ref-type="bibr" rid="B17">Graving et al., 2019</xref>). Individual tracking and pose tracking are closely related. On the one hand, tracking an individual or a local body is always performed to crop active areas of the object to achieve better pose tracking. On the other hand, pose tracking can be regarded as high dimensional individual tracking in some ways, tracking key points of animals (<xref ref-type="bibr" rid="B7">Dell et al., 2014</xref>). Therefore, accurate tracking is the foundation of behavioral analysis, and this is the theoretical basis for SNAP-Tracker to be applied in broader tasks.</p>
<p>Overall, SNAP-Tracker perfectly achieves a balance between applicability and accuracy. The pretrained deep Siamese network makes SNAP-Tracker track the object accurately by comparing the similarity between the template and the target. The model-free tracking strategy equips SNAP-Tracker with broader applicability demonstrated by the experiments above in this article. Strictly speaking, the problem of manual annotation is not thoroughly solved, but the most attractive characteristic of SNAP-Tracker is that it requires only one annotation to start the tracking procedure. Users can correct accidental tracking failures by hands in regular mode or by the detection module in the &#x201C;tracking with detection&#x201D; mode. For the convenience of users, we designed the detection module is in a close loop, in which the tracking module offers elementary results as training data to the detection module, and the latter can help increase the accuracy of tracking. There is still some weakness of the SNAP-Tracker that should be solved to improve further the usability and accuracy of SNAP-Tracker, such as the setting of optimal hyperparameters (<xref ref-type="bibr" rid="B11">Dong et al., 2021</xref>) and the tracking failures when the target is occluded (<xref ref-type="bibr" rid="B12">Dong et al., 2017</xref>). We have considered some of these in the subsequent improvement of SNAP-Tracker.</p>
</sec>
</sec>
<sec id="S5" sec-type="conclusion">
<title>Conclusion</title>
<p>In conclusion, we provide a tracking method in a model-free fashion. Users can easily apply it to various tasks without heavy data annotations. We hope that our tool can lower the barrier to using deep learning methods in animal behavioral analysis and help solve practical tracking problems in related fields.</p>
</sec>
<sec id="S6" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://drive.google.com/file/d/1k0w_lgIBd5xIY0f63J8VfuccvHZ7spsD/view?usp=sharing">https://drive.google.com/file/d/1k0w_lgIBd5xIY0f63J8VfuccvHZ7spsD/view?usp=sharing</ext-link>.</p>
</sec>
<sec id="S7">
<title>Author Contributions</title>
<p>LS, WW, KD, YT, and LM conceived and designed the experiments. LS wrote the code and annotated the dataset. LS and KS performed the experiments. LS, WW, KS, XL, and KD analyzed the results. LS, WW, and KS prepared the figures. LS, WW, YT, and LM wrote and revised the draft. All authors provided comments and approved the manuscript.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S8" sec-type="funding-information">
<title>Funding</title>
<p>This work was partially supported by grants from the National Natural Science Foundation of China under Contract Nos. 62027804, 61825101, and 62088102.</p>
</sec>
<ack>
<p>We would like to thank Jun Ding for providing mouse running videos and valuable comments during the study.</p>
</ack>
<sec id="S10" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fnbeh.2022.759943/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fnbeh.2022.759943/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Image_1.TIF" id="FS1" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_2.TIF" id="FS2" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_3.TIF" id="FS3" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_4.TIF" id="FS4" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_1.WMV" id="VS1" mimetype="audio/x-ms-wmv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_2.MP4" id="VS2" mimetype="audio/x-ms-wmv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_3.MP4" id="VS3" mimetype="audio/x-ms-wmv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_4.MP4" id="VS4" mimetype="audio/x-ms-wmv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_5.MP4" id="VS5" mimetype="audio/x-ms-wmv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_6.MP4" id="VS6" mimetype="audio/x-ms-wmv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video_7.MP4" id="VS7" mimetype="audio/x-ms-wmv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Anderson</surname> <given-names>D. J.</given-names></name> <name><surname>Perona</surname> <given-names>P.</given-names></name></person-group> (<year>2014</year>). <article-title>Toward a science of computational ethology.</article-title> <source><italic>Neuron</italic></source> <volume>84</volume> <fpage>18</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2014.09.005</pub-id> <pub-id pub-id-type="pmid">25277452</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berdahl</surname> <given-names>A.</given-names></name> <name><surname>Torney</surname> <given-names>C. J.</given-names></name> <name><surname>Ioannou</surname> <given-names>C. C.</given-names></name> <name><surname>Faria</surname> <given-names>J. J.</given-names></name> <name><surname>Couzin</surname> <given-names>I. D.</given-names></name></person-group> (<year>2013</year>). <article-title>Emergent sensing of complex environments by mobile animal groups.</article-title> <source><italic>Science</italic></source> <volume>339</volume> <fpage>574</fpage>&#x2013;<lpage>576</lpage>. <pub-id pub-id-type="doi">10.1126/science.1225883</pub-id> <pub-id pub-id-type="pmid">23372013</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bertinetto</surname> <given-names>L.</given-names></name> <name><surname>Valmadre</surname> <given-names>J.</given-names></name> <name><surname>Henriques</surname> <given-names>J. F.</given-names></name> <name><surname>Vedaldi</surname> <given-names>A.</given-names></name> <name><surname>Torr</surname> <given-names>P. H. S.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Fully-Convolutional Siamese Networks for Object Tracking</article-title>&#x201D; in <source><italic>Computer Vision &#x2013; ECCV 2016 Workshops.</italic></source> <role>eds</role> <person-group person-group-type="editor"><name><surname>Hua</surname> <given-names>G.</given-names></name> <name><surname>J&#x00E9;gou</surname> <given-names>H.</given-names></name></person-group> (<publisher-loc>Germany</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>).</citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bromley</surname> <given-names>J.</given-names></name> <name><surname>Guyon</surname> <given-names>I.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>S&#x00E4;ckinger</surname> <given-names>E.</given-names></name> <name><surname>Shah</surname> <given-names>R.</given-names></name></person-group> (<year>1993</year>). <article-title>Signature verification using a&#x201D; siamese&#x201D; time delay neural network.</article-title> <source><italic>Adv. Neural Inform. Process. Syst.</italic></source> <volume>6</volume> <fpage>737</fpage>&#x2013;<lpage>744</lpage>.</citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>R. E.</given-names></name> <name><surname>Stanford</surname> <given-names>L.</given-names></name> <name><surname>Schellinck</surname> <given-names>H. M.</given-names></name></person-group> (<year>2000</year>). <article-title>Developing standardized behavioral tests for knockout and mutant mice.</article-title> <source><italic>ILAR J.</italic></source> <volume>41</volume> <fpage>163</fpage>&#x2013;<lpage>174</lpage>. <pub-id pub-id-type="doi">10.1093/ilar.41.3.163</pub-id> <pub-id pub-id-type="pmid">11406708</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dall</surname> <given-names>S. R.</given-names></name> <name><surname>Bell</surname> <given-names>A. M.</given-names></name> <name><surname>Bolnick</surname> <given-names>D. I.</given-names></name> <name><surname>Ratnieks</surname> <given-names>F. L.</given-names></name></person-group> (<year>2012</year>). <article-title>An evolutionary ecology of individual differences.</article-title> <source><italic>Ecol. Lett.</italic></source> <volume>15</volume> <fpage>1189</fpage>&#x2013;<lpage>1198</lpage>. <pub-id pub-id-type="doi">10.1111/j.1461-0248.2012.01846.x</pub-id> <pub-id pub-id-type="pmid">22897772</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dell</surname> <given-names>A. I.</given-names></name> <name><surname>Bender</surname> <given-names>J. A.</given-names></name> <name><surname>Branson</surname> <given-names>K.</given-names></name> <name><surname>Couzin</surname> <given-names>I. D.</given-names></name> <name><surname>de Polavieja</surname> <given-names>G. G.</given-names></name> <name><surname>Noldus</surname> <given-names>L. P.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Automated image-based tracking and its application in ecology.</article-title> <source><italic>Trends Ecol. Evol.</italic></source> <volume>29</volume> <fpage>417</fpage>&#x2013;<lpage>428</lpage>. <pub-id pub-id-type="doi">10.1016/j.tree.2014.05.004</pub-id> <pub-id pub-id-type="pmid">24908439</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Dong</surname> <given-names>W.</given-names></name> <name><surname>Socher</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Kai</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>F.-F.</given-names></name></person-group> (<year>2009</year>). &#x201C;<article-title>ImageNet: a large-scale hierarchical image database</article-title>&#x201D; in <source><italic>2009 IEEE Conference on Computer Vision and Pattern Recognition.</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>IEEE</publisher-name>). <pub-id pub-id-type="doi">10.1109/TMI.2016.2528162</pub-id> <pub-id pub-id-type="pmid">26886976</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dennis</surname> <given-names>R. L.</given-names></name> <name><surname>Newberry</surname> <given-names>R. C.</given-names></name> <name><surname>Cheng</surname> <given-names>H. W.</given-names></name> <name><surname>Estevez</surname> <given-names>I.</given-names></name></person-group> (<year>2008</year>). <article-title>Appearance matters: artificial marking alters aggression and stress.</article-title> <source><italic>Poult. Sci.</italic></source> <volume>87</volume> <fpage>1939</fpage>&#x2013;<lpage>1946</lpage>. <pub-id pub-id-type="doi">10.3382/ps.2007-00311</pub-id> <pub-id pub-id-type="pmid">18809854</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dewsbury</surname> <given-names>D. A.</given-names></name></person-group> (<year>1992</year>). <article-title>Comparative psychology and ethology: a reassessment.</article-title> <source><italic>Am. Psychol.</italic></source> <volume>47</volume> <fpage>208</fpage>&#x2013;<lpage>215</lpage>. <pub-id pub-id-type="doi">10.1037/0003-066x.47.2.208</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Shao</surname> <given-names>L.</given-names></name> <name><surname>Ling</surname> <given-names>H.</given-names></name> <name><surname>Porikli</surname> <given-names>F.</given-names></name></person-group> (<year>2021</year>). <article-title>Dynamical Hyperparameter Optimization via Deep Reinforcement Learning in Tracking.</article-title> <source><italic>IEEE Transac. Patt. Analy. Mach. Intell.</italic></source> <volume>43</volume> <fpage>1515</fpage>&#x2013;<lpage>1529</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2019.2956703</pub-id> <pub-id pub-id-type="pmid">31796388</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Yu</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>Occlusion-Aware Real-Time Object Tracking.</article-title> <source><italic>IEEE Transac. Multimed.</italic></source> <volume>19</volume> <fpage>763</fpage>&#x2013;<lpage>771</lpage>. <pub-id pub-id-type="doi">10.1109/TMM.2016.2631884</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frye</surname> <given-names>M. A.</given-names></name> <name><surname>Dickinson</surname> <given-names>M. H.</given-names></name></person-group> (<year>2004</year>). <article-title>Closing the loop between neurobiology and flight behavior in Drosophila.</article-title> <source><italic>Curr. Opin. Neurobiol.</italic></source> <volume>14</volume> <fpage>729</fpage>&#x2013;<lpage>736</lpage>. <pub-id pub-id-type="doi">10.1016/j.conb.2004.10.004</pub-id> <pub-id pub-id-type="pmid">15582376</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Girard</surname> <given-names>M. B.</given-names></name> <name><surname>Kasumovic</surname> <given-names>M. M.</given-names></name> <name><surname>Elias</surname> <given-names>D. O.</given-names></name></person-group> (<year>2011</year>). <article-title>Multi-modal courtship in the peacock spider, Maratus volans (OP-Cambridge, 1874).</article-title> <source><italic>PLoS One</italic></source> <volume>6</volume>:<issue>e25390</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0025390</pub-id> <pub-id pub-id-type="pmid">21980440</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gomez-Marin</surname> <given-names>A.</given-names></name> <name><surname>Paton</surname> <given-names>J. J.</given-names></name> <name><surname>Kampff</surname> <given-names>A. R.</given-names></name> <name><surname>Costa</surname> <given-names>R. M.</given-names></name> <name><surname>Mainen</surname> <given-names>Z. F.</given-names></name></person-group> (<year>2014</year>). <article-title>Big behavioral data: psychology, ethology and the foundations of neuroscience.</article-title> <source><italic>Nat. Neurosci.</italic></source> <volume>17</volume>:<issue>1455</issue>. <pub-id pub-id-type="doi">10.1038/nn.3812</pub-id> <pub-id pub-id-type="pmid">25349912</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goodfellow</surname> <given-names>I. J.</given-names></name> <name><surname>Shlens</surname> <given-names>J.</given-names></name> <name><surname>Szegedy</surname> <given-names>C.</given-names></name></person-group> (<year>2014</year>). <article-title>Explaining and harnessing adversarial examples.</article-title> <source><italic>arXiv</italic></source> [preprint]. <volume>arXiv</volume>:<issue>1412.6572</issue>.</citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Graving</surname> <given-names>J. M.</given-names></name> <name><surname>Chae</surname> <given-names>D.</given-names></name> <name><surname>Naik</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Koger</surname> <given-names>B.</given-names></name> <name><surname>Costelloe</surname> <given-names>B. R.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>DeepPoseKit, a software toolkit for fast and robust animal pose estimation using deep learning.</article-title> <source><italic>Elife</italic></source> <volume>8</volume>:<issue>e47994</issue>. <pub-id pub-id-type="doi">10.7554/eLife.47994</pub-id> <pub-id pub-id-type="pmid">31570119</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>J. Z.</given-names></name> <name><surname>Graves</surname> <given-names>A. R.</given-names></name> <name><surname>Guo</surname> <given-names>W. W.</given-names></name> <name><surname>Zheng</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>A.</given-names></name> <name><surname>Rodriguez-Gonzalez</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Cortex commands the performance of skilled movement.</article-title> <source><italic>Elife</italic></source> <volume>4</volume>:<issue>e10774</issue>. <pub-id pub-id-type="doi">10.7554/eLife.10774</pub-id> <pub-id pub-id-type="pmid">26633811</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hattori</surname> <given-names>Y.</given-names></name> <name><surname>Tomonaga</surname> <given-names>M.</given-names></name> <name><surname>Matsuzawa</surname> <given-names>T.</given-names></name></person-group> (<year>2013</year>). <article-title>Spontaneous synchronized tapping to an auditory rhythm in a chimpanzee.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>3</volume>:<issue>1566</issue>. <pub-id pub-id-type="doi">10.1038/srep01566</pub-id> <pub-id pub-id-type="pmid">23535698</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Deep residual learning for image recognition</article-title>&#x201D; in <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Insafutdinov</surname> <given-names>E.</given-names></name> <name><surname>Pishchulin</surname> <given-names>L.</given-names></name> <name><surname>Andres</surname> <given-names>B.</given-names></name> <name><surname>Andriluka</surname> <given-names>M.</given-names></name> <name><surname>Schiele</surname> <given-names>B.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>Deepercut: a deeper, stronger, and faster multi-person pose estimation model</article-title>&#x201D; in <source><italic>European Conference on Computer Vision.</italic></source> (<publisher-loc>Germany</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Keenan</surname> <given-names>W. T.</given-names></name> <name><surname>Rupp</surname> <given-names>A. C.</given-names></name> <name><surname>Ross</surname> <given-names>R. A.</given-names></name> <name><surname>Somasundaram</surname> <given-names>P.</given-names></name> <name><surname>Hiriyanna</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>Z.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>A visual circuit uses complementary mechanisms to support transient and sustained pupil constriction.</article-title> <source><italic>Elife</italic></source> <volume>5</volume>:<issue>e15392</issue>. <pub-id pub-id-type="doi">10.7554/eLife.15392</pub-id> <pub-id pub-id-type="pmid">27669145</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khan</surname> <given-names>A. G.</given-names></name> <name><surname>Sarangi</surname> <given-names>M.</given-names></name> <name><surname>Bhalla</surname> <given-names>U. S.</given-names></name></person-group> (<year>2012</year>). <article-title>Rats track odour trails accurately using a multi-layered strategy with near-optimal sampling.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>3</volume> <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1038/ncomms1712</pub-id> <pub-id pub-id-type="pmid">22426224</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krakauer</surname> <given-names>J. W.</given-names></name> <name><surname>Ghazanfar</surname> <given-names>A. A.</given-names></name> <name><surname>Gomez-Marin</surname> <given-names>A.</given-names></name> <name><surname>MacIver</surname> <given-names>M. A.</given-names></name> <name><surname>Poeppel</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Neuroscience Needs Behavior: correcting a Reductionist Bias.</article-title> <source><italic>Neuron</italic></source> <volume>93</volume> <fpage>480</fpage>&#x2013;<lpage>490</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2016.12.041</pub-id> <pub-id pub-id-type="pmid">28182904</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). <article-title>Imagenet classification with deep convolutional neural networks.</article-title> <source><italic>Adv. Neural Inform. Process. Syst.</italic></source> <volume>25</volume> <fpage>1097</fpage>&#x2013;<lpage>1105</lpage>.</citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Deep learning.</article-title> <source><italic>Nature</italic></source> <volume>521</volume> <fpage>436</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id> <pub-id pub-id-type="pmid">26017442</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Wu</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Zhang</surname> <given-names>F.</given-names></name> <name><surname>Xing</surname> <given-names>J.</given-names></name> <name><surname>Yan</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Siamrpn++: evolution of siamese visual tracking with very deep networks</article-title>&#x201D; in <source><italic>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition.</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Yan</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>W.</given-names></name> <name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>High performance visual tracking with siamese region proposal network</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Maire</surname> <given-names>M.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name> <name><surname>Hays</surname> <given-names>J.</given-names></name> <name><surname>Perona</surname> <given-names>P.</given-names></name> <name><surname>Ramanan</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2014</year>). <source><italic>Microsoft COCO: common Objects in Context.</italic></source> <publisher-loc>Germany</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>.</citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mathis</surname> <given-names>A.</given-names></name> <name><surname>Mamidanna</surname> <given-names>P.</given-names></name> <name><surname>Cury</surname> <given-names>K. M.</given-names></name> <name><surname>Abe</surname> <given-names>T.</given-names></name> <name><surname>Murthy</surname> <given-names>V. N.</given-names></name> <name><surname>Mathis</surname> <given-names>M. W.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>DeepLabCut: markerless pose estimation of user-defined body parts with deep learning.</article-title> <source><italic>Nat. Neurosci.</italic></source> <volume>21</volume> <fpage>1281</fpage>&#x2013;<lpage>1289</lpage>. <pub-id pub-id-type="doi">10.1038/s41593-018-0209-y</pub-id> <pub-id pub-id-type="pmid">30127430</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mathis</surname> <given-names>M. W.</given-names></name> <name><surname>Mathis</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep learning tools for the measurement of animal behavior in neuroscience.</article-title> <source><italic>Curr. Opin. Neurobiol.</italic></source> <volume>60</volume> <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1016/j.conb.2019.10.008</pub-id> <pub-id pub-id-type="pmid">31791006</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mersch</surname> <given-names>D. P.</given-names></name> <name><surname>Crespi</surname> <given-names>A.</given-names></name> <name><surname>Keller</surname> <given-names>L.</given-names></name></person-group> (<year>2013</year>). <article-title>Tracking individuals shows spatial fidelity is a key regulator of ant social organization.</article-title> <source><italic>Science</italic></source> <volume>340</volume> <fpage>1090</fpage>&#x2013;<lpage>1093</lpage>. <pub-id pub-id-type="doi">10.1126/science.1234316</pub-id> <pub-id pub-id-type="pmid">23599264</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nathan</surname> <given-names>R.</given-names></name> <name><surname>Getz</surname> <given-names>W. M.</given-names></name> <name><surname>Revilla</surname> <given-names>E.</given-names></name> <name><surname>Holyoak</surname> <given-names>M.</given-names></name> <name><surname>Kadmon</surname> <given-names>R.</given-names></name> <name><surname>Saltz</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2008</year>). <article-title>A movement ecology paradigm for unifying organismal movement research.</article-title> <source><italic>Proc. Natl. Acad. Sci. U. S. A.</italic></source> <volume>105</volume> <fpage>19052</fpage>&#x2013;<lpage>19059</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0800375105</pub-id> <pub-id pub-id-type="pmid">19060196</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>A.</given-names></name> <name><surname>Yosinski</surname> <given-names>J.</given-names></name> <name><surname>Clune</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). &#x201C;<article-title>Deep neural networks are easily fooled: high confidence predictions for unrecognizable images</article-title>,&#x201D; in <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ohayon</surname> <given-names>S.</given-names></name> <name><surname>Avni</surname> <given-names>O.</given-names></name> <name><surname>Taylor</surname> <given-names>A. L.</given-names></name> <name><surname>Perona</surname> <given-names>P.</given-names></name> <name><surname>Roian Egnor</surname> <given-names>S. E.</given-names></name></person-group> (<year>2013</year>). <article-title>Automated multi-day tracking of marked mice for the analysis of social behaviour.</article-title> <source><italic>J. Neurosci. Methods</italic></source> <volume>219</volume> <fpage>10</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1016/j.jneumeth.2013.05.013</pub-id> <pub-id pub-id-type="pmid">23810825</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ota</surname> <given-names>N.</given-names></name> <name><surname>Gahr</surname> <given-names>M.</given-names></name> <name><surname>Soma</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>Tap dancing birds: the multimodal mutual courtship display of males and females in a socially monogamous songbird.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>5</volume>:<issue>16614</issue>. <pub-id pub-id-type="doi">10.1038/srep16614</pub-id> <pub-id pub-id-type="pmid">26583485</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pereira</surname> <given-names>T. D.</given-names></name> <name><surname>Aldarondo</surname> <given-names>D. E.</given-names></name> <name><surname>Willmore</surname> <given-names>L.</given-names></name> <name><surname>Kislin</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>S. S.</given-names></name> <name><surname>Murthy</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Fast animal pose estimation using deep neural networks.</article-title> <source><italic>Nat. Methods</italic></source> <volume>16</volume> <fpage>117</fpage>&#x2013;<lpage>125</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0234-5</pub-id> <pub-id pub-id-type="pmid">30573820</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pereira</surname> <given-names>T. D.</given-names></name> <name><surname>Shaevitz</surname> <given-names>J. W.</given-names></name> <name><surname>Murthy</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Quantifying behavior to understand the brain.</article-title> <source><italic>Nat. Neurosci.</italic></source> <volume>23</volume> <fpage>1537</fpage>&#x2013;<lpage>1549</lpage>. <pub-id pub-id-type="doi">10.1038/s41593-020-00734-z</pub-id> <pub-id pub-id-type="pmid">33169033</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Perez-Escudero</surname> <given-names>A.</given-names></name> <name><surname>Vicente-Page</surname> <given-names>J.</given-names></name> <name><surname>Hinz</surname> <given-names>R. C.</given-names></name> <name><surname>Arganda</surname> <given-names>S.</given-names></name> <name><surname>de Polavieja</surname> <given-names>G. G.</given-names></name></person-group> (<year>2014</year>). <article-title>idTracker: tracking individuals in a group by automatic identification of unmarked animals.</article-title> <source><italic>Nat. Methods</italic></source> <volume>11</volume> <fpage>743</fpage>&#x2013;<lpage>748</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2994</pub-id> <pub-id pub-id-type="pmid">24880877</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>P&#x00E9;rez-Escudero</surname> <given-names>A.</given-names></name> <name><surname>Vicente-Page</surname> <given-names>J.</given-names></name> <name><surname>Hinz</surname> <given-names>R. C.</given-names></name> <name><surname>Arganda</surname> <given-names>S.</given-names></name> <name><surname>De Polavieja</surname> <given-names>G. G.</given-names></name></person-group> (<year>2014</year>). <article-title>idTracker: tracking individuals in a group by automatic identification of unmarked animals.</article-title> <source><italic>Nat. Methods</italic></source> <volume>11</volume>:<issue>743</issue>. <pub-id pub-id-type="pmid">24880877</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Real</surname> <given-names>E.</given-names></name> <name><surname>Shlens</surname> <given-names>J.</given-names></name> <name><surname>Mazzocchi</surname> <given-names>S.</given-names></name> <name><surname>Pan</surname> <given-names>X.</given-names></name> <name><surname>Vanhoucke</surname> <given-names>V.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>Youtube-boundingboxes: a large high-precision human-annotated data set for object detection in video</article-title>&#x201D; in <source><italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rodriguez</surname> <given-names>A.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Klaminder</surname> <given-names>J.</given-names></name> <name><surname>Brodin</surname> <given-names>T.</given-names></name> <name><surname>Andersson</surname> <given-names>P. L.</given-names></name> <name><surname>Andersson</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>ToxTrac : a fast and robust software for tracking organisms.</article-title> <source><italic>Methods Ecol. Evol.</italic></source> <volume>9</volume> <fpage>460</fpage>&#x2013;<lpage>464</lpage>. <pub-id pub-id-type="doi">10.1111/2041-210x.12874</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Romero-Ferrero</surname> <given-names>F.</given-names></name> <name><surname>Bergomi</surname> <given-names>M. G.</given-names></name> <name><surname>Hinz</surname> <given-names>R. C.</given-names></name> <name><surname>Heras</surname> <given-names>F. J.</given-names></name> <name><surname>de Polavieja</surname> <given-names>G. G.</given-names></name></person-group> (<year>2019</year>). <article-title>Idtracker. ai: tracking all individuals in small or large collectives of unmarked animals.</article-title> <source><italic>Nat. Methods</italic></source> <volume>16</volume> <fpage>179</fpage>&#x2013;<lpage>182</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0295-5</pub-id> <pub-id pub-id-type="pmid">30643215</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Russakovsky</surname> <given-names>O.</given-names></name> <name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Su</surname> <given-names>H.</given-names></name> <name><surname>Krause</surname> <given-names>J.</given-names></name> <name><surname>Satheesh</surname> <given-names>S.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>ImageNet Large Scale Visual Recognition Challenge.</article-title> <source><italic>Int. J. Comput. Vision</italic></source> <volume>115</volume> <fpage>211</fpage>&#x2013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-015-0816-y</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seibenhener</surname> <given-names>M. L.</given-names></name> <name><surname>Wooten</surname> <given-names>M. C.</given-names></name></person-group> (<year>2015</year>). <article-title>Use of the open field maze to measure locomotor and anxiety-like behavior in mice.</article-title> <source><italic>J. Vis. Exp.</italic></source> <volume>6</volume>:<issue>e52434</issue>. <pub-id pub-id-type="doi">10.3791/52434</pub-id> <pub-id pub-id-type="pmid">25742564</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shemesh</surname> <given-names>Y.</given-names></name> <name><surname>Sztainberg</surname> <given-names>Y.</given-names></name> <name><surname>Forkosh</surname> <given-names>O.</given-names></name> <name><surname>Shlapobersky</surname> <given-names>T.</given-names></name> <name><surname>Chen</surname> <given-names>A.</given-names></name> <name><surname>Schneidman</surname> <given-names>E.</given-names></name></person-group> (<year>2013</year>). <article-title>High-order social interactions in groups of mice.</article-title> <source><italic>Elife</italic></source> <volume>2</volume>:<issue>e00759</issue>. <pub-id pub-id-type="doi">10.7554/eLife.00759</pub-id> <pub-id pub-id-type="pmid">24015357</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Snowdon</surname> <given-names>C. T.</given-names></name></person-group> (<year>1983</year>). <article-title>Ethology, Comparative Psychology, and Animal Behavior.</article-title> <source><italic>Annu. Rev. Psychol.</italic></source> <volume>34</volume> <fpage>63</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.ps.34.020183.000431</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sridhar</surname> <given-names>V. H.</given-names></name> <name><surname>Roche</surname> <given-names>D. G.</given-names></name> <name><surname>Gingins</surname> <given-names>S.</given-names></name> <name><surname>B&#x00F6;rger</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>Tracktor: image-based automated tracking of animal movement and behaviour.</article-title> <source><italic>Methods Ecol. Evol.</italic></source> <volume>10</volume> <fpage>815</fpage>&#x2013;<lpage>820</lpage>. <pub-id pub-id-type="doi">10.1111/2041-210x.13166</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>R-SiamNet: rOI-Align Pooling Baesd Siamese Network for Object Tracking</article-title>,&#x201D; in <source><italic>2020 IEEE Conference on Multimedia Information Processing and Retrieval (MIPR).</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Worrall</surname> <given-names>A. D.</given-names></name> <name><surname>Marslin</surname> <given-names>R. F.</given-names></name> <name><surname>Sullivan</surname> <given-names>G. D.</given-names></name> <name><surname>Baker</surname> <given-names>K. D.</given-names></name></person-group> (<year>1991</year>). <source><italic>Model-based Tracking.</italic></source> <publisher-loc>London</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Van Der Maaten</surname> <given-names>L.</given-names></name></person-group> (<year>2013</year>). <article-title>Preserving structure in model-free tracking.</article-title> <source><italic>IEEE Transac. Patt. Analy. Mach. Intell.</italic></source> <volume>36</volume> <fpage>756</fpage>&#x2013;<lpage>769</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2013.221</pub-id> <pub-id pub-id-type="pmid">26353198</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Wu</surname> <given-names>W.</given-names></name> <name><surname>Yan</surname> <given-names>J.</given-names></name> <name><surname>Hu</surname> <given-names>W.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Distractor-aware siamese networks for visual object tracking</article-title>&#x201D; in <source><italic>Proceedings of the European Conference on Computer Vision (ECCV).</italic></source> (<publisher-loc>United States</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation></ref>
</ref-list>
</back>
</article>
