<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Imaging.</journal-id>
<journal-title>Frontiers in Imaging</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Imaging.</abbrev-journal-title>
<issn pub-type="epub">2813-3315</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimag.2024.1387543</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Imaging</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Overhead fisheye cameras for indoor monitoring: challenges and recent progress</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Konrad</surname> <given-names>Janusz</given-names></name>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2601280/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Cokbas</surname> <given-names>Mertcan</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2832800/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tezcan</surname> <given-names>M. Ozan</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2659798/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ishwar</surname> <given-names>Prakash</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2265337/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff><institution>Department of Electrical and Computer Engineering, Boston University</institution>, <addr-line>Boston, MA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Lucio Marcenaro, University of Genoa, Italy</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Pietro Morerio, Italian Institute of Technology (IIT), Italy</p>
<p>Shaobo Liu, Wuhan University of Technology, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Janusz Konrad <email>jkonrad&#x00040;bu.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>27</day>
<month>09</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>3</volume>
<elocation-id>1387543</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>08</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Konrad, Cokbas, Tezcan and Ishwar.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Konrad, Cokbas, Tezcan and Ishwar</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Monitoring the number of people in various spaces of a building is important for optimizing space usage, assisting with public safety, and saving energy. Diverse approaches have been developed for different end goals, from ID card readers for space management, to surveillance cameras for security, to CO<sub>2</sub> sensing for HVAC control. In the last few years, fisheye cameras mounted overhead have become the sensing modality of choice because they offer large-area coverage and significantly-reduced occlusions but research efforts are still nascent. In this paper, we provide an overview of recent research efforts in this area and propose one new direction. First, we identify benefits and challenges related to inference from top-view fisheye images, and summarize key public datasets. Then, we review efforts in algorithm development for detecting people from a single fisheye frame and from a group of sequential frames. Finally, we focus on counting people indoors. While this is straightforward for a single camera, when multiple cameras are used to monitor a space, person re-identification is needed to avoid overcounting. We describe a framework for people counting using two cameras and demonstrate its effectiveness in a large classroom for location-based person re-identification. To support people counting in even larger spaces, we propose two new person re-identification algorithms using <italic>N</italic> &#x0003E; 2 overhead fisheye cameras. We provide ample experimental results throughout the paper.</p></abstract>
<kwd-group>
<kwd>fisheye cameras</kwd>
<kwd>overhead viewpoint</kwd>
<kwd>indoor monitoring</kwd>
<kwd>people detection</kwd>
<kwd>people counting</kwd>
<kwd>person re-identification</kwd>
<kwd>surveillance</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="7"/>
<equation-count count="6"/>
<ref-count count="47"/>
<page-count count="20"/>
<word-count count="12770"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Imaging Applications</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Knowing how many people are in various spaces of a building is important for security/safety, space management, and saving energy. From the security standpoint, it is critical to know where people are in order to ensure everyone is accounted for in an emergency situation (e.g., fire). The recent experience with COVID-19 has shown the importance of understanding occupancy patterns to assure public safety (e.g., monitoring overcrowding). The pandemic has also dramatically impacted the office-building market, leading to new office-usage patterns. A trend of &#x0201C;flexible workspace&#x0201D; is emerging, where desks are not assigned to employees but can be reserved whenever employees return in-person for work, meetings, etc. Real-time, accurate knowledge of workspace occupancy is essential for an effective implementation of this concept. A similar knowledge of where people are is essential in other industries, such as retail, e.g., the number of people visiting a specific store isle or queuing for checkout. Finally, fine-grained occupancy information plays increasingly important role in HVAC control; by matching air flow to occupancy, significant energy savings can be realized compared to binary (on/off) control.</p>
<p>Many approaches have been proposed for occupancy sensing in commercial buildings, such as <italic>active</italic> methods that require carrying a cell-phone or swiping an ID card, and <italic>passive indirect</italic> approaches that use environmental data related to human presence (e.g., CO<sub>2</sub> level, humidity, temperature). However, <italic>passive direct</italic> methods that capture occupants&#x00027; features such as appearance, movement, body heat, etc., are most robust and fine-grained. Unlike active methods, passive direct methods do not require carrying a beacon, and compared to passive indirect methods can provide fine granularity in people counts and their locations.</p>
<p>In this paper, we focus on fisheye cameras mounted overhead to capture appearance and movement of occupants. <xref ref-type="fig" rid="F1">Figure 1</xref> illustrates a potential deployment scenario of fisheye cameras in a large space. Firstly, we review benefits and challenges related to inference from top-view fisheye images. Then, we briefly summarize key image datasets captured indoors by overhead fisheye cameras. Subsequently, we review and evaluate five recent people-detection algorithms on these datasets. Although the best algorithms achieve excellent performance, we observe that under severe occlusions or significant pose changes they <italic>intermittently</italic> fail. To address this, we describe three extensions that leverage spatio-temporal continuity of human motion to improve the detection accuracy. As the space under monitoring increases in size, a single fisheye camera becomes insufficient for accurate detection of people. While a logical solution is to use multiple overhead fisheye cameras, it is unclear how to count and track people who simultaneously appear in the field-of-view (FOV) of <italic>multiple</italic> cameras. In order to resolve this, a person <italic>re-identification</italic> (ReID) algorithm is needed, but very few methods have been proposed for fisheye cameras. We describe a location-based ReID algorithm to match identities between two fisheye cameras and demonstrate its effectiveness in people-counting in a large classroom with highly-dynamic occupancy. To support even larger spaces, we propose two novel extensions to <italic>N</italic>&#x0003E;2 cameras and evaluate them as well. Throughout the paper, we provide numerous experimental results.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Typical monitoring scenario in a large space&#x02014;multiple fisheye cameras are needed resulting in field-of-view overlap and ambiguities in people counting, tracking, etc. (photo by David Iliff. License: CC BY-SA 3.0).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0001.tif"/>
</fig>
</sec>
<sec id="s2">
<title>2 Related work</title>
<sec>
<title>2.1 Fisheye cameras</title>
<p>Unlike standard surveillance cameras, fisheye cameras are equipped with a wide-angle lens. This facilitates wide-area coverage, but leads to various challenges that we discuss in Section 3. Front-facing fisheye cameras have found applications in autonomous navigation, primarily for pedestrian and obstacle detection, and have been widely researched (Cordts et al., <xref ref-type="bibr" rid="B10">2016</xref>; Yogamani et al., <xref ref-type="bibr" rid="B44">2019</xref>; Ye et al., <xref ref-type="bibr" rid="B43">2020</xref>; Liao et al., <xref ref-type="bibr" rid="B24">2023</xref>). Down-facing fisheye cameras, mounted above the scene of interest, have recently emerged as an alternative to standard side-mounted surveillance cameras due to the wide-area coverage and reduced occlusions. While in outdoor scenarios this is very much limited by the ability to mount such cameras above the scene (e.g., lamp posts), in indoor scenarios the mounting is relatively straightforward (e.g., suspension from the ceiling).</p>
</sec>
<sec>
<title>2.2 People detection</title>
<p>People detection in images from standard surveillance cameras has rich literature spanning at least two decades, from classical methods applying SVM classification to Histogram of Oriented Gradients (HOG) features (Dalal and Triggs, <xref ref-type="bibr" rid="B12">2005</xref>) or using AdaBoost classifier with Aggregate Channel Features (ACF) (Doll&#x000E1;r et al., <xref ref-type="bibr" rid="B14">2014</xref>) to more recent deep-learning methods such as YOLO (Redmon et al., <xref ref-type="bibr" rid="B30">2016</xref>), SSD (Liu et al., <xref ref-type="bibr" rid="B26">2016</xref>) and R-CNN (Girshick, <xref ref-type="bibr" rid="B20">2015</xref>; Ren et al., <xref ref-type="bibr" rid="B32">2015</xref>). However, such methods directly applied to top-view fisheye images perform poorly due to a dramatic range of viewpoints in the same image (people under the camera are seen from above, but those farther away are seen from a side perspective) and arbitrary body orientations (e.g., standing people appear radially in images and can be seen &#x0201C;upside-down&#x0201D;) as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. Furthermore, although more subtle, lens distortions cause body-shape deformations, especially close to fisheye-image periphery, which also penalizes person-detection performance. We discuss these issues in more detail in Section 3.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Test spaces where most of the training/testing data were collected using Axis M3057-PLVE (3,072 &#x000D7; 2,048 pixels, 360&#x000B0; &#x000D7; 185&#x000B0; lens) fisheye cameras mounted 3.15 m above the floor, indicated by red arrows: <bold>(A)</bold> conference room with single camera (where HABBOF was recorded); <bold>(B)</bold> classroom with single camera (where CEPDOF was recorded); <bold>(C)</bold> large classroom with three cameras (where FRIDA was recorded). Corresponding typical occupancy scenarios are shown in the second row <bold>(D&#x02013;F)</bold>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0002.tif"/>
</fig>
<p>In the last decade, person-detection methods have been developed specifically for top-view fisheye images. Early attempts focused on model-based feature extraction and various adaptations to account for fisheye geometry. In perhaps the first work, background subtraction was combined with a probabilistic body-appearance model and followed by Kernel Ridge Regression (Saito et al., <xref ref-type="bibr" rid="B34">2011</xref>). Chiang and Wang (<xref ref-type="bibr" rid="B5">2014</xref>) rotated each fisheye image in small angular steps and applied SVM to HOG features extracted from the top-center part of the image to detect people. Krams and Kiryati (<xref ref-type="bibr" rid="B22">2017</xref>) applied standard ACF classifier to dewarped features extracted from a fisheye image. Demirkus et al. (<xref ref-type="bibr" rid="B13">2017</xref>) also used ACF to learn different-size models dependent on the distance from image center.</p>
<p>The most recent methods are CNN-based end-to-end algorithms. Seidel et al. (<xref ref-type="bibr" rid="B35">2018</xref>) applied YOLO to dewarped versions of overlapping windows extracted from a fisheye image, but tested the algorithm on a private dataset only. Tamura et al. (<xref ref-type="bibr" rid="B36">2019</xref>) introduced a rotation-invariant version of YOLO, that was trained on rotated images from COCO 2017 (Lin et al., <xref ref-type="bibr" rid="B25">2014</xref>), however the inference stage assumed that bounding boxes are aligned with the image radius thus not allowing for arbitrary body orientations. Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) rotated each fisheye image in 15&#x000B0; steps and applied YOLOv3 (Redmon and Farhadi, <xref ref-type="bibr" rid="B31">2018</xref>) to the top-center part of the image where people usually appear upright, followed by post-processing to remove multiple detections of the same person. They also proposed an extension in which the algorithm is applied only to changed areas, as determined by background subtraction, rather than to the whole image. Minh et al. (<xref ref-type="bibr" rid="B28">2021</xref>) proposed an anchor-free CNN that allows bounding-box rotation and speeds up the inference. Chiang et al. (<xref ref-type="bibr" rid="B6">2021</xref>) proposed to unwrap patches from a fisheye image using simple fisheye-lens model in order to compose a perspective image for inference by YOLO, followed by post-processing to remove duplicate detections. Wei et al. (<xref ref-type="bibr" rid="B41">2022</xref>) applied a CNN with deformable convolution kernels to account for geometric distortions in top-view fisheye images, however tested the approach only on their own dataset. Finally, Tamura and Yoshinaga (<xref ref-type="bibr" rid="B37">2023</xref>) extended their earlier work by training on rectilinear datasets while leveraging ground-truth <italic>segmentations</italic> to fit bounding boxes more tightly around human bodies. Since deep-learning algorithms require extensive training data, a number of top-view fisheye-image datasets aimed at people detection have been published; we discuss the most commonly-used ones in Section 4.</p>
</sec>
<sec>
<title>2.3 Person re-identification</title>
<p>Person re-identification is a key component of people counting and by itself is a vast research area of critical importance for visual surveillance. While a detailed review of methods proposed for standard surveillance cameras is beyond the scope of this paper, below we briefly summarize key challenges and types of methods proposed.</p>
<p>Traditional person ReID is concerned with retrieving a person of interest across multiple cameras with <italic>non-overlapping</italic> FOVs. In <italic>closed-world</italic> person ReID, typically a single visual modality is used (e.g., RGB), person detections (bounding boxes) are assumed known and reliable, and the query person appears in the gallery set. There exist many methods developed in this context but they, in general, include three components: feature representation, metric learning and ranking optimization. <italic>Open-world</italic> person ReID attempts to address real-world challenges, such as multiple data modalities (e.g., RGB, depth, text), end-to-end ReID without pre-computed person detections (direct ReID from images/videos), semi-supervised or unsupervised learning with limited/unavailable annotations, dealing with noisy annotations, or open-set ReID when correct match is missing from the gallery. Two recent surveys by Ye et al. (<xref ref-type="bibr" rid="B42">2022</xref>) and by Zhang et al. (<xref ref-type="bibr" rid="B46">2024</xref>) discuss dozens of methods proposed and include experimental comparisons.</p>
<p>However, person ReID methods developed for rectilinear cameras perform poorly on top-view fisheye images, although re-training on fisheye data somewhat improves performance (Cokbas et al., <xref ref-type="bibr" rid="B8">2022</xref>). In addition to arbitrary body orientations (e.g., &#x0201C;upside-down&#x0201D;), dramatic viewpoint differences (e.g., from above in one camera view, but from side-perspective in another camera view) and lens-distortions (more significant when a person appears at image periphery), another challenge for person ReID is body-scale difference between cameras. Traditional person ReID was developed for cameras with non-overlapping FOVs. Since each camera is oriented toward its own area of interest, people appearing in areas monitored by different cameras will often appear at a reasonably-similar size. This is not the case for fisheye ReID considered here. Since the overhead fisheye cameras have overlapping FOVs and since ReID is performed on images captured at the <italic>same</italic> time instant, a person might be under one camera but far away from another camera. In addition to different viewpoints, there will be a dramatic difference in person-image size and geometric distortion (person under a camera will be very large and seen from above, while person far away will be tiny and geometrically-distorted). For a detailed discussion and examples, please see Cokbas (<xref ref-type="bibr" rid="B7">2023</xref>) and Cokbas et al. (<xref ref-type="bibr" rid="B9">2023</xref>).</p>
<p>Very few methods have been proposed to date for person ReID using overhead fisheye cameras. The earliest work by Barman et al. (<xref ref-type="bibr" rid="B1">2018</xref>) considers only matching people who are located at a similar distance from each camera, thus assuring similar body size (and, potentially, similar viewpoint). Another work by Blott et al. (<xref ref-type="bibr" rid="B3">2019</xref>) uses tracking to extract three distinct viewpoints (back, side, front) that are subsequently jointly matched between cameras, which is similar in spirit to multiple-shot person ReID developed for standard cameras by Bazzani et al. (<xref ref-type="bibr" rid="B2">2010</xref>). However, this approach requires reliable tracking and visibility of each person from three very different angles&#x02014;neither can be guaranteed. Both works report results only on private datasets. Another work somehwat related to person ReID is on tracking people in overhead fisheye views by Wang and Chiang (<xref ref-type="bibr" rid="B40">2023</xref>). The authors use their own person detection method (Chiang et al., <xref ref-type="bibr" rid="B6">2021</xref>) and then apply a variant of DeepSORT for tracking.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Overhead fisheye cameras: benefits and challenges</title>
<sec>
<title>3.1 Benefits</title>
<sec>
<title>3.1.1 Wide field of view</title>
<p>The key advantage of fisheye cameras over their rectilinear counterparts is their wide FOV resulting from a particular lens design. The fisheye FOV covers 360&#x000B0; in plane parallel to the sensor and 165&#x02013;200&#x000B0; orthogonally. In contrast, a typical surveillance camera equipped with rectilinear lens covers 60&#x02013;100&#x000B0; in horizontal and vertical dimensions of the sensor. Suspended from the ceiling, a fisheye camera can effectively monitor large area (depending on the installation height). In comparison to rectilinear cameras, fewer fisheye cameras are usually needed to monitor a space thus reducing system complexity and cost. <xref ref-type="fig" rid="F2">Figures 2A</xref>&#x02013;<xref ref-type="fig" rid="F2">C</xref> show a conference room and two classrooms with fisheye cameras suspended from the ceiling, in which our datasets were recorded (Section 4). <xref ref-type="fig" rid="F2">Figures 2D</xref>&#x02013;<xref ref-type="fig" rid="F2">F</xref> show images from these cameras from typical testing. Clearly, people detection in such images faces several challenges.</p>
</sec>
<sec>
<title>3.1.2 Reduced occlusions</title>
<p>In addition to the wide FOV, the overhead camera mounting significantly reduces the severity of occlusions (by other people or furniture), which can be seen in <xref ref-type="fig" rid="F2">Figures 2D</xref>&#x02013;<xref ref-type="fig" rid="F2">F</xref>. However, the overhead viewpoint results in certain challenges, discussed below.</p>
</sec>
</sec>
<sec>
<title>3.2 Challenges</title>
<sec>
<title>3.2.1 Circular field of view</title>
<p>In rectilinear-lens cameras, an RGB sensor records only the central rectangular portion of the FOV and the lens is carefully designed to project straight lines in the physical world onto straight lines on the sensor surface. However, in fisheye cameras only straight lines in the physical world that belong to a plane orthogonal to the sensor and pass through its center result in straight lines in the fisheye image; other lines are curved (e.g., horizontal edges of whiteboards in <xref ref-type="fig" rid="F2">Figures 2D</xref>, <xref ref-type="fig" rid="F2">E</xref>). This geometric distortion introduced by the fisheye lens also affects human-body shape, especially if a person is not in an upright position, thus posing a challenge for both person detection and ReID.</p>
</sec>
<sec>
<title>3.2.2 Non-linear foreshortening</title>
<p>In order to capture a wide FOV, the fisheye lens is designed in a particular way that introduces radial distortions (non-linearity) in the captured images. For example, the projection of a person standing directly under the camera (e.g., certain shoulder width), becomes smaller at 3 m away from the camera and even smaller at 6 m away. This size compression is <italic>linear</italic> in standard cameras thanks to rectilinear lens, but in fisheye cameras the doubling of the distance from the camera results in size compression by more than 2, especially pronounced close to FOV periphery. This non-linear mapping of physical distances (and of body sizes) poses challenges for person detection and ReID in fisheye images.</p>
</sec>
<sec>
<title>3.2.3 Overhead viewpoint</title>
<p>It may result in unusual human-body appearance; people directly under the camera are seen from above (<xref ref-type="fig" rid="F2">Figures 2E</xref>, <xref ref-type="fig" rid="F2">F</xref>) but those farther away are seen from a side-view perspective. This dramatic viewpoint variability is not encountered in images captured by a side-mounted rectilinear camera. Another important consequence of the overhead viewpoint is that <italic>standing</italic> people appear in radial directions in a fisheye image, including horizontal and &#x0201C;upside-down&#x0201D; orientations. In fact, people can appear at <italic>any</italic> orientation in overhead fisheye images. This is unlike in images captured by side-mounted rectilinear cameras, where standing people appear upright and for which the vast majority of people-detection algorithms have been developed (bounding boxes aligned with image axes).</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Overhead fisheye datasets</title>
<p>In order to develop people-detection algorithms for overhead fisheye cameras, annotated image datasets are needed for performance evaluation and, potentially, algorithm training. While very large fisheye-image datasets have been collected in front-facing scenarios for autonomous navigation (Cordts et al., <xref ref-type="bibr" rid="B10">2016</xref>; Yogamani et al., <xref ref-type="bibr" rid="B44">2019</xref>; Ye et al., <xref ref-type="bibr" rid="B43">2020</xref>; Liao et al., <xref ref-type="bibr" rid="B24">2023</xref>), few and much smaller fisheye-image datasets are available in overhead scenario intended for indoor surveillance. A recent survey by Yu et al. (<xref ref-type="bibr" rid="B45">2023</xref>) describes 16 natural and synthetic datasets collected with top-view fisheye cameras, but most of them either focus on action recognition, or contain very few frames, or do not provide full-body bounding boxes. In <xref ref-type="table" rid="T1">Table 1</xref>, we summarize a subset of these datasets that are composed of natural, as opposed to synthetic, overhead images and annotated with full-body bounding boxes either aligned with image axes or rotated.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Recent image datasets captured by overhead fisheye cameras in various venues and occupancy scenarios, annotated with full-body bounding boxes either aligned with image axes or rotated.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Venue</bold></th>
<th valign="top" align="center"><bold>Num. of videos / frames</bold></th>
<th valign="top" align="center"><bold>Resolution</bold></th>
<th valign="top" align="center"><bold>Max. num. of people</bold></th>
<th valign="top" align="center"><bold>B-box alingment</bold></th>
<th valign="top" align="center"><bold>Challenges</bold></th>
<th valign="top" align="center"><bold>Sample image</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Mirror Worlds (MW)<sup>a</sup></td>
<td valign="top" align="center">Hallways, medium-size rooms</td>
<td valign="top" align="center">30 / 13k</td>
<td valign="top" align="center">1&#x02013;2 MP</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">Axis</td>
<td valign="top" align="center">Walking, sitting</td>
<td valign="top" align="center"><xref ref-type="fig" rid="F3">Figures 3A</xref>, <xref ref-type="fig" rid="F3">B</xref></td>
</tr>
<tr>
<td valign="top" align="left">Mirror Worlds -Rotated (MW-R)<sup>b</sup></td>
<td valign="top" align="center">Hallways, medium-size rooms</td>
<td valign="top" align="center">19 / 8,752</td>
<td valign="top" align="center">1&#x02013;2 MP</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">Rotated</td>
<td valign="top" align="center">Walking, sitting</td>
<td valign="top" align="center"><xref ref-type="fig" rid="F3">Figures 3A</xref>, <xref ref-type="fig" rid="F3">B</xref></td>
</tr>
<tr>
<td valign="top" align="left">Human-Aligned Bounding Boxes from Overhead Fisheye Cameras (HABBOF)<sup>c</sup></td>
<td valign="top" align="center">Computer lab, conference room</td>
<td valign="top" align="center">4 / 5,837</td>
<td valign="top" align="center">4.2 MP</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">Rotated</td>
<td valign="top" align="center">Walking, sitting, varying illumination</td>
<td valign="top" align="center"><xref ref-type="fig" rid="F2">Figure 2D</xref></td>
</tr>
<tr>
<td valign="top" align="left">Challenging Events for Person Detection from Overhead Fisheye Images (CEPDOF)<sup>d</sup></td>
<td valign="top" align="center">Classroom</td>
<td valign="top" align="center">8 / 25,504</td>
<td valign="top" align="center">1.2&#x02013;4.2 MP</td>
<td valign="top" align="center">13</td>
<td valign="top" align="center">Rotated</td>
<td valign="top" align="center">Crowded, occlusions, rare poses, camouflage, people on a screen, low light</td>
<td valign="top" align="center"><xref ref-type="fig" rid="F2">Figures 2E</xref>, <xref ref-type="fig" rid="F3">3C</xref>&#x02013;<xref ref-type="fig" rid="F3">F</xref></td>
</tr>
<tr>
<td valign="top" align="left">In-the-Wild Events for People Detection and Tracking from Overhead Fisheye Cameras (WEPDTOF)<sup>e</sup></td>
<td valign="top" align="center">Varying, from YouTube</td>
<td valign="top" align="center">16 / 10,544</td>
<td valign="top" align="center">0.6&#x02013;5 MP</td>
<td valign="top" align="center">35</td>
<td valign="top" align="center">Rotated</td>
<td valign="top" align="center">Crowded, occlusions, camouflage, distorted FOV, varying illumination</td>
<td valign="top" align="center"><xref ref-type="fig" rid="F3">Figures 3G</xref>, <xref ref-type="fig" rid="F3">H</xref></td>
</tr>
<tr>
<td valign="top" align="left">Fisheye Re-Identification Dataset with Annotations (FRIDA)<sup>f</sup></td>
<td valign="top" align="center">Large classroom</td>
<td valign="top" align="center">4 / 18,318 3 cameras</td>
<td valign="top" align="center">4.2 MP</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center">Rotated</td>
<td valign="top" align="center">Crowded, occlusions, rare poses, far away people</td>
<td valign="top" align="center"><xref ref-type="fig" rid="F2">Figure 2F</xref></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>See a survey by Yu et al. (<xref ref-type="bibr" rid="B45">2023</xref>) for additional fisheye-image datasets.</p>
<p><sup>a</sup><ext-link ext-link-type="uri" xlink:href="https://www2.icat.vt.edu/mirrorworlds/challenge/index.html">https://www2.icat.vt.edu/mirrorworlds/challenge/index.html</ext-link></p>
<p><sup>b</sup><ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/projects/vsns/cossy/datasets/mw-r">https://vip.bu.edu/projects/vsns/cossy/datasets/mw-r</ext-link></p>
<p><sup>c</sup><ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/projects/vsns/cossy/datasets/habbof">https://vip.bu.edu/projects/vsns/cossy/datasets/habbof</ext-link></p>
<p><sup>d</sup><ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/projects/vsns/cossy/datasets/cepdof">https://vip.bu.edu/projects/vsns/cossy/datasets/cepdof</ext-link></p>
<p><sup>e</sup><ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/projects/vsns/cossy/datasets/wepdtof">https://vip.bu.edu/projects/vsns/cossy/datasets/wepdtof</ext-link></p>
<p><sup>f</sup><ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/projects/vsns/cossy/datasets/frida">https://vip.bu.edu/projects/vsns/cossy/datasets/frida</ext-link>.</p>
</table-wrap-foot>
</table-wrap>
<p>These datasets are essential for training modern data-driven person detection and ReID algorithms so that they can learn how to handle various challenges, such as dramatic viewpoint changes, arbitrary body orientations, geometric body-shape distortions and dramatic body size (scale) differences (ReID), as already discussed in Sections 2, 3.</p>
</sec>
<sec id="s5">
<title>5 Finding people in overhead fisheye images</title>
<p>All people-detection algorithms for overhead fisheye cameras discussed in Section 2 perform inference for each video frame separately. In the next section, we show that while the performance of such algorithms has steadily improved on &#x0201C;staged&#x0201D; datasets, when applied to real-life videos it significantly degrades. In the subsequent section, we show that by leveraging the spatio-temporal continuity of human motion, the detection performance can be significantly improved.</p>
<sec>
<title>5.1 Detection using a single video frame</title>
<p>Many recent approaches use RAPiD (Rotation-Aware People Detection) (Duan et al., <xref ref-type="bibr" rid="B16">2020</xref>) as a benchmark for performance evaluation. RAPiD is a CNN based on YOLOv3 (Redmon and Farhadi, <xref ref-type="bibr" rid="B31">2018</xref>) adapted to accommodate bounding-box rotation and trained using the original YOLOv3 loss function augmented with a novel periodic loss for angle regression. RAPiD handles unusual body viewpoints by training on a variety of fisheye images and its source code is publicly available<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref>. <xref ref-type="table" rid="T2">Table 2</xref> compares performance of four recent algorithms against RAPiD. All algorithms were pre-trained on COCO 2017 (Lin et al., <xref ref-type="bibr" rid="B25">2014</xref>) and then fine-tuned and tested on MW-R, HABBOF and CEPDOF via cross-dataset validation. Specificallly, two datasets were used for training and the third one was used for testing, and then the roles were swapped. This resulted in three sets of performance measures that were averaged and are shown in <xref ref-type="table" rid="T2">Table 2</xref>. The two versions of RAPiD differ in training/testing image resolutions to allow comparison with other methods. Even at the lower resolution, RAPiD significantly outperforms other methods in all performance metrics. However, ARPD by Minh et al. (<xref ref-type="bibr" rid="B28">2021</xref>) offers much faster inference than RAPiD at the cost of accuracy.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Performance of recent single-frame people-detection algorithms on MW-R, HABBOF and CEPDOF via 3-fold cross-dataset validation.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Algorithm</bold></th>
<th valign="top" align="center"><bold>Image resolution</bold></th>
<th valign="top" align="center"><bold>AP<sub>50</sub>&#x02191;(<italic>%</italic>)</bold></th>
<th valign="top" align="center"><bold>Precision &#x02191;</bold></th>
<th valign="top" align="center"><bold>Recall &#x02191;</bold></th>
<th valign="top" align="center"><bold>F-score &#x02191;</bold></th>
<th valign="top" align="center"><bold>Run time [sec]</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Tamura et al. (<xref ref-type="bibr" rid="B36">2019</xref>)</td>
<td valign="top" align="center">608 &#x000D7; 608</td>
<td valign="top" align="center">75.5</td>
<td valign="top" align="center">0.906</td>
<td valign="top" align="center">0.704</td>
<td valign="top" align="center">0.778</td>
<td valign="top" align="center">0.098</td>
</tr>
<tr>
<td valign="top" align="left">Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) AB</td>
<td valign="top" align="center">1,024 &#x000D7; 1,024</td>
<td valign="top" align="center">88.7</td>
<td valign="top" align="center">0.887</td>
<td valign="top" align="center">0.844</td>
<td valign="top" align="center">0.849</td>
<td valign="top" align="center">1.776</td>
</tr>
<tr>
<td valign="top" align="left">Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) AA</td>
<td valign="top" align="center">1,024 &#x000D7; 1,024</td>
<td valign="top" align="center">83.3</td>
<td valign="top" align="center">0.919</td>
<td valign="top" align="center">0.775</td>
<td valign="top" align="center">0.816</td>
<td valign="top" align="center">1.477</td>
</tr>
<tr>
<td valign="top" align="left">Duan et al. (<xref ref-type="bibr" rid="B16">2020</xref>) RAPiD</td>
<td valign="top" align="center">608 &#x000D7; 608</td>
<td valign="top" align="center">92.1</td>
<td valign="top" align="center"><bold>0.952</bold></td>
<td valign="top" align="center">0.862</td>
<td valign="top" align="center">0.897</td>
<td valign="top" align="center">0.118</td>
</tr>
<tr>
<td valign="top" align="left">Duan et al. (<xref ref-type="bibr" rid="B16">2020</xref>) RAPiD</td>
<td valign="top" align="center">1,024 &#x000D7; 1,024</td>
<td valign="top" align="center"><bold>93.5</bold></td>
<td valign="top" align="center">0.932</td>
<td valign="top" align="center"><bold>0.903</bold></td>
<td valign="top" align="center"><bold>0.913</bold></td>
<td valign="top" align="center">0.223</td>
</tr>
<tr>
<td valign="top" align="left">Minh et al. (<xref ref-type="bibr" rid="B28">2021</xref>) ARPD</td>
<td valign="top" align="center">512 &#x000D7; 512</td>
<td valign="top" align="center">90.5</td>
<td valign="top" align="center">0.931</td>
<td valign="top" align="center">0.845</td>
<td valign="top" align="center">0.884</td>
<td valign="top" align="center"><bold>0.066</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>All metrics are averaged over three splits, so the F-measure is not equal to the harmonic mean of Precision and Recall. The AB (activity-blind) algorithm by Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) applies YOLOv3 to all rotated windows in the frame, while the AA (activity-aware) variant limits YOLOv3 to windows overlapping areas of change obtained from background subtraction. ARPD values are averages obtained from the original paper by Minh et al. (<xref ref-type="bibr" rid="B28">2021</xref>). The average run times per image are obtained on NVIDIA Tesla V100 GPU except for ARPD measured on NVIDIA GTX 1070 Ti. The best performance and lowest run time are shown in boldface.</p>
</table-wrap-foot>
</table-wrap>
<p>We would like to point out that although the test datasets consist of top-view fisheye images, challenges vary (<xref ref-type="table" rid="T1">Table 1</xref>). While in MW-R and HABBOF people are either standing or walking, CEPDOF is more challenging with many unusual poses, severe occlusions and low-light conditions. <xref ref-type="fig" rid="F3">Figures 3A</xref>&#x02013;<xref ref-type="fig" rid="F3">F</xref> show sample detections produced by RAPiD under various challenges. Except for extreme cases (people on the screen, low light), RAPiD performs exceedingly well. This is confirmed by AP<sub>50</sub> which exceeds 93%, and Precision, Recall and F-score that are over 0.9. However, all three datasets were recorded using high-quality cameras in &#x0201C;staged&#x0201D; scenarios (controlled environment, subjects instructed to behave in a certain way). Would these algorithms perform equally well &#x0201C;in the wild&#x0201D;, that is in uncontrolled real-life situations?</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Examples of people detections by RAPiD on videos with a wide range of challenges: <bold>(A)</bold> different body poses (MW-R); <bold>(B)</bold> person directly under camera (MW-R); <bold>(C)</bold> various body angles (CEPDOF); <bold>(D)</bold> occlusions (CEPDOF); <bold>(E)</bold> people visible on the screen (CEPDOF); <bold>(F)</bold> low-light scenario (CEPDOF); <bold>(G)</bold> severe camouflage (WEPDTOF); and <bold>(H)</bold> tiny moving people (WEPDTOF). Green boxes are true positives, red boxes are false positives, and yellow boxes are false negatives.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0003.tif"/>
</fig>
<p><xref ref-type="table" rid="T3">Table 3</xref> shows performance of four of these algorithms<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> on the WEPDTOF dataset collected from YouTube, that includes real-life challenges as detailed in Section 4. In addition to AP<sub>50</sub>, similarly to the MS COCO challenge (Lin et al., <xref ref-type="bibr" rid="B25">2014</xref>), we report this metric for small, medium and large bounding boxes denoted as <inline-formula><mml:math id="M7"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, <inline-formula><mml:math id="M8"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, and <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, respectively. Clearly, RAPiD outperforms other algorithms in terms of AP<sub>50</sub> but is outperformed by AA and AB in terms of Precision and F-score. This is largely due to the fact that AA and AB compute bounding-box predictions from overlapped crops of a rotated image and combine these results in a post-processing step. Thus, they analyze a person&#x00027;s appearance multiple times, each at a slightly-different rotation angle, which boosts the confidence score of the bounding box for that person but hugely increases the complexity (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Performance of recent single-frame people-detection algorithms on WEPDTOF.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Algorithm</bold></th>
<th valign="top" align="center"><bold>AP<sub>50</sub>&#x02191;(%)</bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M1"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mi>&#x02191;</mml:mi></mml:math></inline-formula>(%)</bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M2"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mi>&#x02191;</mml:mi></mml:math></inline-formula>(%)</bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M3"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup><mml:mi>&#x02191;</mml:mi></mml:math></inline-formula>(%)</bold></th>
<th valign="top" align="center"><bold>Precision &#x02191;</bold></th>
<th valign="top" align="center"><bold>Recall &#x02191;</bold></th>
<th valign="top" align="center"><bold>F-score &#x02191;</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Tamura et al. (<xref ref-type="bibr" rid="B36">2019</xref>)</td>
<td valign="top" align="center">59.8</td>
<td valign="top" align="center">11.6</td>
<td valign="top" align="center">65.2</td>
<td valign="top" align="center">61.3</td>
<td valign="top" align="center">0.777</td>
<td valign="top" align="center">0.508</td>
<td valign="top" align="center">0.581</td>
</tr>
<tr>
<td valign="top" align="left">Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) AB</td>
<td valign="top" align="center">69.8</td>
<td valign="top" align="center">15.8</td>
<td valign="top" align="center">71.3</td>
<td valign="top" align="center">63.1</td>
<td valign="top" align="center"><bold>0.818</bold></td>
<td valign="top" align="center">0.643</td>
<td valign="top" align="center">0.702</td>
</tr>
<tr>
<td valign="top" align="left">Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) AA</td>
<td valign="top" align="center">68.3</td>
<td valign="top" align="center">11.4</td>
<td valign="top" align="center">70.1</td>
<td valign="top" align="center">63.7</td>
<td valign="top" align="center">0.804</td>
<td valign="top" align="center">0.647</td>
<td valign="top" align="center"><bold>0.705</bold></td>
</tr>
<tr>
<td valign="top" align="left">Duan et al. (<xref ref-type="bibr" rid="B16">2020</xref>) RAPiD</td>
<td valign="top" align="center"><bold>72.0</bold></td>
<td valign="top" align="center"><bold>18.4</bold></td>
<td valign="top" align="center"><bold>72.8</bold></td>
<td valign="top" align="center"><bold>67.9</bold></td>
<td valign="top" align="center">0.731</td>
<td valign="top" align="center"><bold>0.676</bold></td>
<td valign="top" align="center">0.668</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><inline-formula><mml:math id="M4"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, <inline-formula><mml:math id="M5"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M6"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are AP<sub>50</sub> values for small (area &#x02264; 1,200), medium (1,200 &#x0003C; area &#x02264; 8,000) and large (8,000 &#x0003C; area) bounding boxes, with areas normalized to image size of 1,024 &#x000D7; 1,024. The best performance is shown in boldface.</p>
</table-wrap-foot>
</table-wrap>
<p>Note that all metrics in <xref ref-type="table" rid="T3">Table 3</xref> are significantly lower than those in <xref ref-type="table" rid="T2">Table 2</xref> due to challenges captured in WEPDTOF. When visually evaluating these results by playing video with superimposed detections, we observed that some bounding boxes produced by RAPiD appear intermittently (they disappear for a frame or two but then reappear again) thus contributing to the reduced performance. Visual examples of misses (false negatives) in two challenging sequences from WEPDTOF are shown in <xref ref-type="fig" rid="F3">Figures 3G</xref>, <xref ref-type="fig" rid="F3">H</xref>.</p>
</sec>
<sec>
<title>5.2 Detection using a group of video frames</title>
<p>To address the intermittent behavior of detections and improve performance, some form of temporal coherence of detections should be incorporated into the people-detection algorithm since people are either static or move incrementally in space-time. In this context, we developed three extensions of RAPiD (Tezcan et al., <xref ref-type="bibr" rid="B38">2022</xref>) each leveraging temporal information differently, that we briefly summarize below. The source code for these algorithms is publicly available<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref>.</p>
<sec>
<title>5.2.1 RAPiD &#x0002B; REPP</title>
<p>This method first applies RAPiD to individual frames and then revises the detections by applying post-processing based on Robust and Efficient Post-Processing (REPP) algorithm proposed by Sabater et al. (<xref ref-type="bibr" rid="B33">2020</xref>). Since REPP produces axis-aligned bounding boxes, it was modified to account for bounding-box rotations. In the training step, a similarity function is computed from annotated data using the following features from <italic>pairs</italic> of bounding boxes in consecutive frames: Euclidean distance between their centers, ratio of their widths, ratio of their heights, absolute difference between their angles, and Intersection over Union (IoU) between them. In the inference step, first bounding boxes are <italic>detected</italic> by RAPiD and linked between consecutive frames into &#x0201C;tubelets&#x0201D; using similarity scores computed by the trained similarity function. Then, the confidence score, location, size and angle of the bounding boxes in each &#x0201C;tubelet&#x0201D; are smoothed out. This results in more temporally-consistent bounding-box characteristics potentially leading to more consistent behavior in time.</p>
</sec>
<sec>
<title>5.2.2 RAPiD &#x0002B; (FG)FA</title>
<p>This is an end-to-end approach that extends RAPiD by integrating information from a group of neighboring video frames to stabilize intermittent detections. The integration mechanism was inspired by Flow-Guided Feature Aggregation (FGFA) proposed by Zhu et al. (<xref ref-type="bibr" rid="B47">2017</xref>). More specifically, since RAPiD is a YOLO-based algorithm, it first extracts feature maps from the input image at three resolutions. In order to integrate temporal information, at each resolution level feature maps from 10 past frames and 10 future frames are warped to the current feature map by means of motion compensation and all of them are linearly combined. Unlike in Zhu et al. (<xref ref-type="bibr" rid="B47">2017</xref>), RAPiD &#x0002B; FGFA uses the Farneb&#x000E4;ck algorithm (Farneb&#x000E4;ck, <xref ref-type="bibr" rid="B18">2003</xref>) to compute optical flow since it outperforms FlowNet (Dosovitskiy et al., <xref ref-type="bibr" rid="B15">2015</xref>) on overhead fisheye videos. The aggregated feature maps are then transformed into bounding-box-related feature maps, and based on them the detection head predicts bounding boxes. RAPiD &#x0002B; FA is a simplified version of RAPiD &#x0002B; FGFA and applies feature aggregation with adaptive weights but without motion compensation.</p>
<p><xref ref-type="table" rid="T4">Table 4</xref> shows performance of the three multi-frame detection algorithms described above against single-frame algorithms from <xref ref-type="table" rid="T3">Table 3</xref> on the WEPDTOF dataset. The multi-frame algorithms significantly outperform single-frame algorithms in terms of all AP<sub>50</sub> metrics. Interestingly, the AB algorithm by Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) (YOLOv3 applied to all rotated windows) again achieves the highest Precision although at the cost of very high computational complexity. This is due to overlapping windows resulting in multiple detections of the same person that pruned by subsequent post-processing rarely results in a false positive. Among the multi-frame algorithms, RAPiD &#x0002B; REPP turns out to be very complex computationally. The other two algorithms are about 3 times more complex than the one by Tamura et al. (<xref ref-type="bibr" rid="B36">2019</xref>) but offer over 15% points boost in AP<sub>50</sub>. <xref ref-type="fig" rid="F4">Figure 4</xref> shows people-detection examples produced by RAPiD and three multi-frame algorithms for three challenging video sequences from WEPDTOF. While RAPiD&#x0002B;REPP corrects one false positive and one false negative, and RAPiD &#x0002B; FA corrects four false negatives and one false positive, it also introduces one false positive. However, RAPiD &#x0002B; FGFA corrects four false negatives and one false positive without introducing any errors.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Performance of three multi-frame people-detection algorithms (Tezcan et al., <xref ref-type="bibr" rid="B38">2022</xref>) against recent single-frame algorithms on WEPDTOF. The average run times per image are obtained on NVIDIA Tesla V100 GPU.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Algorithm</bold></th>
<th valign="top" align="center"><bold>AP<sub>50</sub> (%)</bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M10"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> (%)</bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> (%)</bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M12"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">AP</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>50</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> (%)</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F-score</bold></th>
<th valign="top" align="center"><bold>Run time (s)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Single- frame</td>
<td valign="top" align="center">Tamura et al. (<xref ref-type="bibr" rid="B36">2019</xref>)</td>
<td valign="top" align="center">59.8</td>
<td valign="top" align="center">11.6</td>
<td valign="top" align="center">65.2</td>
<td valign="top" align="center">61.3</td>
<td valign="top" align="center">0.777</td>
<td valign="top" align="center">0.508</td>
<td valign="top" align="center">0.581</td>
<td valign="top" align="center"><bold>0.098</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) AB</td>
<td valign="top" align="center">69.8</td>
<td valign="top" align="center">15.8</td>
<td valign="top" align="center">71.3</td>
<td valign="top" align="center">63.1</td>
<td valign="top" align="center"><bold>0.818</bold></td>
<td valign="top" align="center">0.643</td>
<td valign="top" align="center">0.702</td>
<td valign="top" align="center">1.776</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Li et al. (<xref ref-type="bibr" rid="B23">2019</xref>) AA</td>
<td valign="top" align="center">68.3</td>
<td valign="top" align="center">11.4</td>
<td valign="top" align="center">70.1</td>
<td valign="top" align="center">63.7</td>
<td valign="top" align="center">0.804</td>
<td valign="top" align="center">0.647</td>
<td valign="top" align="center">0.705</td>
<td valign="top" align="center">1.477</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Duan et al. (<xref ref-type="bibr" rid="B16">2020</xref>) RAPiD</td>
<td valign="top" align="center">72.0</td>
<td valign="top" align="center">18.4</td>
<td valign="top" align="center">72.8</td>
<td valign="top" align="center">67.9</td>
<td valign="top" align="center">0.731</td>
<td valign="top" align="center">0.676</td>
<td valign="top" align="center">0.668</td>
<td valign="top" align="center">0.118</td>
</tr>
<tr>
<td valign="top" align="left">Multi- frame</td>
<td valign="top" align="center">RAPiD &#x0002B; REPP</td>
<td valign="top" align="center">73.7</td>
<td valign="top" align="center">19.8</td>
<td valign="top" align="center">74.2</td>
<td valign="top" align="center">70.2</td>
<td valign="top" align="center">0.794</td>
<td valign="top" align="center">0.679</td>
<td valign="top" align="center">0.703</td>
<td valign="top" align="center">1.667</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">RAPiD &#x0002B; FA</td>
<td valign="top" align="center">75.6</td>
<td valign="top" align="center">19.6</td>
<td valign="top" align="center">77.5</td>
<td valign="top" align="center">71.8</td>
<td valign="top" align="center">0.784</td>
<td valign="top" align="center">0.672</td>
<td valign="top" align="center">0.689</td>
<td valign="top" align="center">0.269</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">RAPiD &#x0002B; FGFA</td>
<td valign="top" align="center"><bold>76.6</bold></td>
<td valign="top" align="center"><bold>20.9</bold></td>
<td valign="top" align="center"><bold>77.9</bold></td>
<td valign="top" align="center"><bold>72.0</bold></td>
<td valign="top" align="center">0.803</td>
<td valign="top" align="center"><bold>0.691</bold></td>
<td valign="top" align="center"><bold>0.725</bold></td>
<td valign="top" align="center">0.300</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best performance and lowest run time are shown in boldface.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Qualitative results of RAPiD and its multi-frame extensions on videos from WEPDTOF. Green boxes are true positives, red boxes are false positives, and yellow boxes are false negatives.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0004.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s6">
<title>6 Counting people using overhead fisheye cameras</title>
<sec>
<title>6.1 Counting metrics</title>
<p>If a people-detection algorithm, such as those discussed in Section 5, is perfectly accurate, then counting people is as simple as counting bounding boxes. However, certain errors in people detection may still result in a correct people count, for example when a false positive and false negative occur in the same image. These two errors cancel each other, so an accurate metric for people counting should ignore such scenarios. Furthermore, we are interested in how far off is an estimated count from a true count. In this section, we use the following metrics to evaluate performance of people-counting algorithms:</p>
<disp-formula id="E1"><mml:math id="M13"><mml:mtable columnalign="right"><mml:mtr><mml:mtd><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>X</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle mathvariant="double-struck"><mml:mn>1</mml:mn></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>&#x02264;</mml:mo><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B7;<sub><italic>i</italic></sub> and <inline-formula><mml:math id="M14"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the true and estimated people counts in frame number <italic>i</italic>, <italic>M</italic> is the total number of frames and <italic>1</italic>(&#x003B5;) is an indicator function, that is <italic>1</italic>(&#x003B5;) equals 1 if &#x003B5; is true and 0 otherwise. While the Mean Absolute Error (<italic>MAE</italic>) is a commonly-used metric, it is not meaningful when comparing algorithms at different occupancy levels (e.g., 80 people vs. 8 people). This is addressed by the Mean Absolute Error per person (<italic>MAE</italic><sub><italic>pp</italic></sub>) which divides <italic>MAE</italic> by the average occupancy over <italic>M</italic> frames. The X-Accuracy (<italic>Acc</italic><sub><italic>X</italic></sub>) quantifies people-counting performance as &#x0201C;accuracy with slack of <italic>X</italic>&#x0201D;. For <italic>X</italic> &#x0003D; 0 this definition reverts to the traditional definition of accuracy, but for larger values of <italic>X</italic> it tolerates the departure of <inline-formula><mml:math id="M15"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> from &#x003B7;<sub><italic>i</italic></sub> by up to <italic>X</italic>. For example, <italic>Acc</italic><sub>5</sub> gives the percentage of frames in which the estimated count is within 5 of the true count.</p>
</sec>
<sec>
<title>6.2 Dataset</title>
<p>To evaluate performance of various algorithms, we recorded data over 3 days in a large classroom (<xref ref-type="fig" rid="F2">Figure 2C</xref>) equipped with three cameras. On day 1 there were 11 high-occupancy periods (lectures) with up to 87 occupants (<xref ref-type="fig" rid="F2">Figure 2F</xref>), on day 2 there were four such periods with up to 65 occupants, while on day 3 the classroom was mostly empty with maximum occupancy of 9 for a short period of time. We annotated all frames in terms of the number of people in the classroom, but <italic>not</italic> in terms of bounding boxes.</p>
</sec>
<sec>
<title>6.3 Counting people using one camera</title>
<p><xref ref-type="table" rid="T5">Table 5</xref> shows the people-counting performance of RAPiD on this dataset for each of the three cameras. The much larger values of <italic>MAE</italic> on days 1 and 2 are due to high average occupancy on these two days. <italic>MAE</italic><sub><italic>pp</italic></sub>, on the other hand, is similar across all days confirming its relative independence of occupancy scenarios. Its value of about 0.4 suggests RAPiD commits an error of about 40% per person which is high. This mediocre performance is confirmed by the values of <italic>Acc</italic><sub><italic>X</italic></sub>. Cumulatively over 3 days, RAPiD produces exact counts in 46&#x02013;55% of frames depending on the camera and only in 76&#x02013;79% frames with count error of up to 10. Clearly, a single camera is incapable of accurate counting in this large a space.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Single-camera people-counting performance of RAPiD in a 3-day test in large classroom (<xref ref-type="fig" rid="F2">Figure 2C</xref>) equipped with three cameras. The last column shows cumulative metrics computed over 3 days.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Camera</bold></th>
<th valign="top" align="center"><bold>Day 1</bold></th>
<th valign="top" align="center"><bold>Day 2</bold></th>
<th valign="top" align="center"><bold>Day 3</bold></th>
<th valign="top" align="center"><bold>Cumulative</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Average number of people</td>
<td/>
<td valign="top" align="center">29.5</td>
<td valign="top" align="center">20.6</td>
<td valign="top" align="center">0.99</td>
<td valign="top" align="center">16.4</td>
</tr>
<tr>
<td valign="top" align="left"><italic>MAE&#x02193;</italic></td>
<td valign="top" align="center">&#x00023;1</td>
<td valign="top" align="center">11.82</td>
<td valign="top" align="center">6.92</td>
<td valign="top" align="center">0.38</td>
<td valign="top" align="center">6.11</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">&#x00023;2</td>
<td valign="top" align="center">11.60</td>
<td valign="top" align="center">8.55</td>
<td valign="top" align="center">0.48</td>
<td valign="top" align="center">6.61</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">&#x00023;3</td>
<td valign="top" align="center">12.32</td>
<td valign="top" align="center">7.36</td>
<td valign="top" align="center">0.62</td>
<td valign="top" align="center">6.49</td>
</tr>
<tr>
<td valign="top" align="left"><italic>MAE</italic><sub><italic>pp</italic></sub>&#x02193;</td>
<td valign="top" align="center">&#x00023;1</td>
<td valign="top" align="center">0.400</td>
<td valign="top" align="center">0.336</td>
<td valign="top" align="center">0.384</td>
<td valign="top" align="center">0.373</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">&#x00023;2</td>
<td valign="top" align="center">0.393</td>
<td valign="top" align="center">0.415</td>
<td valign="top" align="center">0.480</td>
<td valign="top" align="center">0.404</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">&#x00023;3</td>
<td valign="top" align="center">0.417</td>
<td valign="top" align="center">0.358</td>
<td valign="top" align="center">0.623</td>
<td valign="top" align="center">0.397</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Acc</italic><sub><italic>X</italic></sub> [%] &#x02191;<italic>X</italic>=0/5/10</td>
<td valign="top" align="center">&#x00023;1</td>
<td valign="top" align="center">45/53/59</td>
<td valign="top" align="center">43/64/70</td>
<td valign="top" align="center">74/100/100</td>
<td valign="top" align="center">55/73/77</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">&#x00023;2</td>
<td valign="top" align="center">47/54/60</td>
<td valign="top" align="center">29/60/65</td>
<td valign="top" align="center">67/100/100</td>
<td valign="top" align="center">48/72/76</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">&#x00023;3</td>
<td valign="top" align="center">38/59/64</td>
<td valign="top" align="center">42/64/70</td>
<td valign="top" align="center">55/100/100</td>
<td valign="top" align="center">46/75/79</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>6.4 Counting people using two cameras</title>
<p>Increasing the number of cameras to 2 creates a problem of potential overcounting since the same person may be captured by both cameras due to their wide fields of view (<xref ref-type="fig" rid="F1">Figure 1</xref>). In order to make sure that each person is counted only once, <italic>person re-identification</italic> is needed.</p>
<sec>
<title>6.4.1 Person re-identification between two cameras</title>
<p>Traditional PRID considers scenarios where images of people have been recorded by cameras with non-overlapping fields of view at widely-varying times. For example, some cameras may be mounted at entrances to an airport terminal, another group of cameras may be placed at security checkpoints, and yet another group&#x02014;at the boarding gates. Images of people captured at the entrances and security checkpoints are assumed known and form the gallery set. Similarly, images of people captured at the boarding gates are assumed known and form the query set. The traditional person ReID attempts to match identities between these two sets. For each identity in the query set, the goal is to find all matching identities in the gallery set.</p>
<p>However, the ReID scenario we consider here is different since multiple cameras with <italic>overlapping</italic> fields of view <italic>simultaneously</italic> monitor a space (<xref ref-type="fig" rid="F1">Figure 1</xref>). Person ReID for the purpose of people counting can be performed in two steps: detect people in each camera view at the <italic>same</italic> time instant and then use their appearance (images) to match identities. Identities present in the view of one camera are considered to be the query set and those in the view of another camera are considered to be the gallery set. Therefore, a query identity can match <italic>at most</italic> one identity in the gallery set or none at all, in the case of occlusion or failed person detection, which is an example of the open-world scenario. Note, that in this work we consider frame-to-frame identity matching (single shot), however it would be interesting to extend this to multiple-shot fisheye ReID similarly to a method proposed for rectilinear cameras by Bazzani et al. (<xref ref-type="bibr" rid="B2">2010</xref>). However, this would require reliable tracking in overhead fisheye cameras, a topic still in its infancy.</p>
<p>Traditional person ReID methods rely on appearance of a person, such as color, hand-crafted features or deep-learning features, but they tend to be unreliable for overhead fisheye images since a person&#x00027;s appearance and size dramatically differ depending on this person&#x00027;s location in the room (see <xref ref-type="fig" rid="F2">Figure 2</xref>) as demonstrated in Cokbas (<xref ref-type="bibr" rid="B7">2023</xref>) and Cokbas et al. (<xref ref-type="bibr" rid="B9">2023</xref>). However, in our scenario of simultaneous image capture by cameras with overlapping fields of view, a person can be also re-identified based on their location. Since the cameras are fixed, a person appearing in a camera&#x00027;s FOV appears at a <italic>specific</italic> location of another camera&#x00027;s FOV; this location depends on intrinsic camera parameters, installation height, distance between cameras, etc. A ReID method based on this idea was developed by Bone et al. (<xref ref-type="bibr" rid="B4">2021</xref>) and shown to be very effective. This method requires camera calibration, which we summarize next. Then, we describe key ReID steps shown in a high-level block diagram in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Block diagram of location-based person re-identification (Bone et al., <xref ref-type="bibr" rid="B4">2021</xref>). Locations of people detected by RAPiD in image from camera <italic>&#x00023;K</italic> (blue <italic>K</italic><sub><italic>i</italic></sub>&#x00027;s) and those predicted from camera <italic>&#x00023;L</italic> (red <italic>L</italic><sub><italic>i</italic></sub>&#x00027;s) are used to compute 2-D score matrix <italic><bold>D</bold></italic><sub><italic>KL</italic></sub> using either the PPD or CBD distance metric. An analogous score matrix is computed for locations of people detected in camera <italic>&#x00023;L</italic> image (<italic><bold>D</bold></italic><sub><italic>LK</italic></sub>). A greedy algorithm applied to matrix <inline-formula><mml:math id="M16"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> matches identities (yellow table).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0005.tif"/>
</fig>
<p>For a pair of identical and level ceiling-mounted fisheye cameras (<italic>&#x00023;K</italic> and <italic>&#x00023;L</italic>), this method uses five intrinsic parameters [2-D scaling factor, 2-D optical center offset, and a scalar parameter of the <italic>unified spherical model</italic> for fisheye cameras (Geyer and Danilidis, <xref ref-type="bibr" rid="B19">2001</xref>; Courbon et al., <xref ref-type="bibr" rid="B11">2012</xref>)] and two extrinsic parameters (distance between cameras and relative rotation angle between them). The distance between cameras was precisely measured using a laser measure, but the remaining five parameters were estimated using images of the test space with lights off when rolling a cart with a spherical red LED light mounted at a fixed height. This allowed us to record precise projection of the LED light on two fisheye images at the same time instant. Such pairs of projections are related through a bidirectional mapping (<italic>&#x00023;K</italic>&#x02192;<italic>&#x00023;L</italic> or <italic>&#x00023;L</italic>&#x02192;<italic>&#x00023;K</italic>) which is a function of the intrinsic and extrinsic parameters (Bone et al., <xref ref-type="bibr" rid="B4">2021</xref>). In order to estimate the intrinsic parameters and rotation angle, the Euclidean distance between 1,000&#x0002B; projection pairs was minimized using stochastic gradient descent by sequentially iterating through the two bidirectional mappings. This calibration has to be performed only once for a given camera model (intrinsic parameters). The extrinsic parameters (camera installation height, distance between cameras, rotation angle, etc.) need to be measured or calibrated with each camera-layout change or when adding new cameras. An interesting direction of research would be to develop an unsupervised approach in such cases, similarly to adaptive ReID proposed by Panda et al. (<xref ref-type="bibr" rid="B29">2017</xref>) for open-world dynamic networks of rectilinear cameras.</p>
<p>During ReID (<xref ref-type="fig" rid="F5">Figure 5</xref>), first RAPiD is applied to same-time images from cameras <italic>&#x00023;K</italic> and <italic>&#x00023;L</italic> to detect people; the center of each detected bounding box marks a person&#x00027;s location (blue <italic>K</italic><sub><italic>i</italic></sub>&#x00027;s and <italic>L</italic><sub><italic>i</italic></sub>&#x00027;s). Then, using the intrinsic and extrinsic parameters, and the average height of a person (168 cm), the detected locations in each camera view (blue symbols) are mapped to <italic>predict</italic> these locations in the other camera view (red symbols). Ideally, the predicted locations should coincide with the detected ones, but in reality this is not the case due to imperfect camera model and calibration. In practice, the closer a predicted location (red) is to a detected location (blue) the more likely it is that these are locations of the same person. Bone et al. (<xref ref-type="bibr" rid="B4">2021</xref>) proposed four different distance metrics to quantify the proximity of a predicted location to detected locations in a given view. The fastest metric to compute is called Point-to-Point Distance (PPD). It calculates the Euclidean distance between each predicted location and each detected location to form a 2-D score matrix (<xref ref-type="fig" rid="F5">Figure 5</xref>). The best-performing metric in their tests is called Count-Based Distance (CBD). Unlike PPD, which assumes average height of a person and is not accurate for very tall and short individuals, CBD considers a range of human heights (150&#x02013;190 cm in 2 cm increments). For each detected location in one view, it produces 21 predicted locations in the other view (corresponding to different heights of a person). Then, for each detected location the number of predicted locations (out of 21) for which this detected location is <italic>closest</italic> establishes a count. This count is subtracted from the total number of considered person-heights (21) to establish a distance measure; the smaller the measure (the larger the count) the more likely it is that the detected and predicted locations have the same identity. This metric is computed for each detected and predicted identity to form a 2-D score matrix. Score matrix <italic><bold>D</bold></italic><sub><italic>KL</italic></sub> (<xref ref-type="fig" rid="F5">Figure 5</xref>) is computed for the detections in camera <italic>&#x00023;K</italic> and predictions from camera <italic>&#x00023;L</italic>, while score matrix <italic><bold>D</bold></italic><sub><italic>LK</italic></sub> is computed for the detections in camera <italic>&#x00023;L</italic> and predictions from camera <italic>&#x00023;K</italic>. Finally, to perform bidirectional identity matching a combined score matrix <inline-formula><mml:math id="M17"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is used in a greedy fashion. First, the smallest entry in <italic><bold>S</bold></italic><sub><italic>KL</italic></sub> is found, the corresponding identities are considered a match, and their row and column are removed from the matrix reducing its size by 1 in each dimension. This process is repeated until no more matches are possible.</p>
<p>This location-based approach to person ReID was shown to outperform appearance-based methods by a large margin (over 11% points in mAP for the PPD metric and over 14% points for the CBD metric) on the FRIDA dataset (Cokbas et al., <xref ref-type="bibr" rid="B9">2023</xref>), and is our method of choice for people counting discussed next.</p>
</sec>
<sec>
<title>6.4.2 Removal of double-counts</title>
<p>The ReID method discussed in the previous section is essential for accurate people counting using two cameras. Let <inline-formula><mml:math id="M18"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M19"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> be the estimated people counts in frame number <italic>i</italic> from cameras <italic>&#x00023;K</italic> and <italic>&#x00023;L</italic>, respectively, for example obtained by counting bounding boxes detected by a person-detection algorithm, such as RAPiD. Let <inline-formula><mml:math id="M20"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> be the number of people detections successfully <italic>re-identified</italic> between these two frames. The people-count estimate for this pair of frames is then computed as follows:</p>
<disp-formula id="E2"><label>(1)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the subtraction of <inline-formula><mml:math id="M22"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> removes double counts discovered by person re-identification.</p>
</sec>
<sec>
<title>6.4.3 Experimental results for two cameras</title>
<p><xref ref-type="table" rid="T6">Table 6</xref> shows a 2-camera people-counting performance of RAPiD (to compute <inline-formula><mml:math id="M23"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M24"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> in each frame pair) followed by location-based person re-identification with either the PPD or CBD distance metric (to compute <inline-formula><mml:math id="M25"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>). Compared to <xref ref-type="table" rid="T5">Table 5</xref>, both cumulative <italic>MAE</italic> and <italic>MAE</italic><sub><italic>pp</italic></sub> are reduced about 4 times for both distance metrics. The CBD metric performs slightly better than PPD achieving cumulative <italic>MAE</italic> of 1.59 and <italic>MAE</italic><sub><italic>pp</italic></sub> of 0.097. This is a huge performance improvement over the single-camera results. In particular, the <italic>MAE</italic><sub><italic>pp</italic></sub> value suggests that the two-camera approach commits an error of less than 10% per person compared to 40% for single camera. While the cumulative X-Accuracy for <italic>X</italic> &#x0003D; 0 is slightly reduced compared to <xref ref-type="table" rid="T5">Table 5</xref>, the one for <italic>X</italic> &#x0003D; 5 is improved to 92% and one for <italic>X</italic> &#x0003D; 10 is 98%. Clearly, using two cameras in a large space significantly improves the people-counting accuracy of RAPiD.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Two-camera people-counting performance of RAPiD in a 3-day test in large classroom using cameras &#x00023;2 and &#x00023;3 (<xref ref-type="fig" rid="F2">Figure 2C</xref>) and location-based person-re-identification using the PPD or CBD distance-error metric.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>ReID metric</bold></th>
<th valign="top" align="center"><bold>Day 1</bold></th>
<th valign="top" align="center"><bold>Day 2</bold></th>
<th valign="top" align="center"><bold>Day 3</bold></th>
<th valign="top" align="center"><bold>Cumulative</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Average number of people</td>
<td/>
<td valign="top" align="center">29.5</td>
<td valign="top" align="center">20.6</td>
<td valign="top" align="center">0.99</td>
<td valign="top" align="center">16.4</td>
</tr>
<tr>
<td valign="top" align="left"><italic>MAE&#x02193;</italic></td>
<td valign="top" align="center">PPD</td>
<td valign="top" align="center"><bold>2.42</bold></td>
<td valign="top" align="center">1.74</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">1.63</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">CBD</td>
<td valign="top" align="center">2.43</td>
<td valign="top" align="center"><bold>1.71</bold></td>
<td valign="top" align="center"><bold>0.75</bold></td>
<td valign="top" align="center"><bold>1.59</bold></td>
</tr>
<tr>
<td valign="top" align="left"><italic>MAE</italic><sub><italic>pp</italic></sub>&#x02193;</td>
<td valign="top" align="center">PPD</td>
<td valign="top" align="center">0.082</td>
<td valign="top" align="center">0.084</td>
<td valign="top" align="center">0.849</td>
<td valign="top" align="center">0.100</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">CBD</td>
<td valign="top" align="center">0.082</td>
<td valign="top" align="center"><bold>0.083</bold></td>
<td valign="top" align="center"><bold>0.752</bold></td>
<td valign="top" align="center"><bold>0.097</bold></td>
</tr>
<tr>
<td valign="top" align="left"><italic>Acc</italic><sub><italic>X</italic></sub> [%] &#x02191;<italic>X</italic>=0/5/10</td>
<td valign="top" align="center">PPD</td>
<td valign="top" align="center">41/84/96</td>
<td valign="top" align="center"><bold>36</bold>/<bold>92</bold>/99</td>
<td valign="top" align="center"><bold>50</bold>/<bold>98</bold>/100</td>
<td valign="top" align="center"><bold>42</bold>/92/98</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">CBD</td>
<td valign="top" align="center">41/84/96</td>
<td valign="top" align="center">37/93/99</td>
<td valign="top" align="center">51/99/100</td>
<td valign="top" align="center">43/92/98</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The last column shows cumulative metrics computed across all days. The better performance is shown in boldface.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s7">
<title>7 Counting people in large spaces using <italic>N</italic> &#x0003E; 2 overhead fisheye cameras</title>
<p>The results in Section 6 indicate that two overhead fisheye cameras are sufficient for quite accurate people counting in a 187 m<sup>2</sup> space (<xref ref-type="fig" rid="F2">Figure 2C</xref>). However, in larger spaces more cameras would be needed to maintain a similar level of performance. This would require person ReID between more than two fisheye cameras, a task <italic>unexplored</italic> to-date. In this section, we propose two novel approaches to accomplish this and demonstrate their performance for people counting using three overhead fisheye cameras.</p>
<sec>
<title>7.1 General person ReID based on <italic>N</italic>-dimensional score matrix</title>
<p>We propose a <italic>general</italic> approach to person ReID using <italic>N</italic>-D score matrices. Such matrices can quantify similarity between identities using people locations, as proposed by Bone et al. (<xref ref-type="bibr" rid="B4">2021</xref>), or their appearance, as explored by Cokbas et al. (<xref ref-type="bibr" rid="B9">2023</xref>). We first consider a 3-camera setup (<italic>N</italic> &#x0003D; 3) of our test classroom shown in <xref ref-type="fig" rid="F2">Figure 2C</xref>, however we note that, in general, cameras need not be collinear. In <xref ref-type="fig" rid="F6">Figure 6</xref>, we graphically illustrate (Venn diagram) the general relationship between sets of person detections from three cameras. In this diagram, <italic>C</italic><sub>1</sub>, <italic>C</italic><sub>2</sub>, <italic>C</italic><sub>3</sub> denote the sets of people detections in images simultaneously captured by cameras &#x00023;1, &#x00023;2 and &#x00023;3, respectively. The Venn diagram allows us to compute the actual people count &#x003B7; as follows:</p>
<disp-formula id="E3"><label>(2)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:mi>&#x003B7;</mml:mi><mml:mo>=</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>-</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02229;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>-</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02229;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>-</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02229;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02229;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02229;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where |<italic>C</italic>| denotes the cardinality of set <italic>C</italic>. Clearly, |<italic>C</italic><sub>1</sub>|, |<italic>C</italic><sub>2</sub>|, |<italic>C</italic><sub>3</sub>| are people counts in respective camera views provided by a people-detection algorithm (e.g., RAPiD). The numbers of identities matched between two cameras, namely |<italic>C</italic><sub>1</sub>&#x02229;<italic>C</italic><sub>2</sub>|, |<italic>C</italic><sub>1</sub>&#x02229;<italic>C</italic><sub>3</sub>|, |<italic>C</italic><sub>2</sub>&#x02229;<italic>C</italic><sub>3</sub>|, are provided by a two-camera ReID algorithm, such as the one described in Section 6.4.1. However, we still need to identify the number of identities matched across all three cameras: |<italic>C</italic><sub>1</sub>&#x02229;<italic>C</italic><sub>2</sub>&#x02229;<italic>C</italic><sub>3</sub>|. This necessitates a 3-camera ReID algorithm.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Venn diagram graphically illustrating people-counting scenario for <italic>N</italic> &#x0003D; 3 cameras.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0006.tif"/>
</fig>
<p>As discussed in Section 6.4.1 and shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, person ReID between two cameras results in a 2-D score (distance) matrix <italic><bold>D</bold></italic> that is subject to a greedy search to match identities. With three cameras, this matrix would become 3-dimensional with each entry containing, for example, a measure of appearance similarity for a <italic>triplet</italic> of people detections (one detection from each camera view), or a distance metric computed from locations of this triplet. Rather than defining a new 3-camera distance metric, we adopt 2-camera metrics developed by Bone et al. (<xref ref-type="bibr" rid="B4">2021</xref>), apply them to 3 pairs of cameras and average the resulting scores as follows:</p>
<disp-formula id="E4"><label>(3)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>123</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>12</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>13</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>23</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic><bold>S</bold></italic><sub>12</sub>, <italic><bold>S</bold></italic><sub>13</sub>, <italic><bold>S</bold></italic><sub>23</sub> are 2-D score matrices (like those in <xref ref-type="fig" rid="F5">Figure 5</xref>), <italic><bold>S</bold></italic><sub>123</sub> is a 3-D score matrix and <italic>i</italic><sub><italic>K</italic></sub> is the identity detected in view from camera <italic>&#x00023;K</italic>. Clearly, <italic><bold>S</bold></italic><sub>12</sub>(<italic>i</italic><sub>1</sub>, <italic>i</italic><sub>2</sub>) quantifies the location mismatch for identities <italic>i</italic><sub>1</sub> and <italic>i</italic><sub>2</sub> from cameras &#x00023;1 and &#x00023;2, respectively, and <bold>S</bold><sub>123</sub>(<italic>i</italic><sub>1</sub>, <italic>i</italic><sub>2</sub>, <italic>i</italic><sub>3</sub>) represents the location mismatch for identities <italic>i</italic><sub>1</sub>, <italic>i</italic><sub>2</sub>, <italic>i</italic><sub>3</sub>, each from its respective camera.</p>
<p>An extension to <italic>N</italic>&#x0003E;3 is relatively straightforward but one must carefully consider various combinations of <italic>n</italic> out of <italic>N</italic> cameras, across which identities need to be matched. For example, for <italic>N</italic> &#x0003D; 4 cameras, re-identifications between 2, 3, or 4 camera views are needed in order to obtain the correct overall count. For <italic>N</italic> cameras, the total number of camera combinations to be considered is:</p>
<disp-formula id="E5"><label>(4)</label><mml:math id="M29"><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mtable><mml:mtr><mml:mtd><mml:mi>N</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>n</mml:mi></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle><mml:mo>=</mml:mo><mml:msup><mml:mn>2</mml:mn><mml:mi>N</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>N</mml:mi><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where the summation starts at <italic>n</italic> &#x0003D; 2 since re-identification requires at least two camera views. For <italic>N</italic> &#x0003D; 3, this amounts to four camera combinations which is consistent with four intersections in the Venn diagram in <xref ref-type="fig" rid="F6">Figure 6</xref>. For <italic>N</italic> &#x0003D; 4, there are 11 camera combinations (6 two-camera combinations, 4 three-camera combination, and 1 four-camera combination), and for <italic>N</italic> &#x0003D; 5 there are 26 camera combinations, rapidly increasing with a growing number of cameras.</p>
<p>Clearly, score matrices <bold>S</bold> of up to <italic>N</italic> dimensions are needed for <italic>N</italic>-camera re-identification. One possibility is to generalize <xref ref-type="disp-formula" rid="E3">Equation (2)</xref> to <italic>N</italic> dimensions through the use of the well-known <italic>inclusion-exclusion principle</italic> by van Lint and Wilson (<xref ref-type="bibr" rid="B39">1992</xref>) as follows:</p>
<disp-formula id="E6"><label>(5)</label><mml:math id="M30"><mml:mrow><mml:msub><mml:mstyle mathvariant='bold-italic'><mml:mi>S</mml:mi></mml:mstyle><mml:mrow><mml:mn>12</mml:mn><mml:mo>&#x02026;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;..</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mtable><mml:mtr><mml:mtd><mml:mi>N</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>2</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mstyle mathvariant='bold-italic'><mml:mi>S</mml:mi></mml:mstyle><mml:mrow><mml:mi>k</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>i</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>We would like to emphasize that this approach to <italic>N</italic>-camera person ReID is general and can be applied to both appearance- and location-based features. However, in the remainder of this section we focus on location-based matching due to its superior performance for overhead fisheye cameras (Cokbas et al., <xref ref-type="bibr" rid="B9">2023</xref>) and low computational complexity. The computational complexity is a serious concern for large <italic>N</italic> since the number of camera combinations that need to be considered grows exponentially with a growing <italic>N</italic> <xref ref-type="disp-formula" rid="E5">(4)</xref>.</p>
</sec>
<sec>
<title>7.2 Person ReID based on clustering of real-world locations</title>
<p>The approach we proposed in the previous section is general and applies to both appearance- and location-based features. If applied to location-based features, it effectively performs identity matching based on locations in the 2-D image plane (pixel coordinates); identities are matched based on the proximity of detected and predicted locations in a given camera view. All the predicted locations are obtained by first inverse-mapping 2-D locations detected in one camera view to 3-D coordinates and then forward-mapping these 3-D coordinates to 2-D locations in the other camera view (bidirectional location mapping in <xref ref-type="fig" rid="F5">Figure 5</xref>). This may result in location-error imbalance since the detected locations do not undergo any mapping, while the predicted locations are obtained via inverse-forward mapping that uses intrinsic and extrinsic parameter estimates<xref ref-type="fn" rid="fn0004"><sup>4</sup></xref>, unlikely to be perfectly accurate.</p>
<p>In order to avoid this location-error imbalance, we propose to perform identity matching in real-world coordinates, that is map person-detection locations in <italic>all</italic> camera views to 3-D coordinates. An explanation is required at this point. The inverse mapping from a 2-D location in a camera view to 3-D coordinates must be constrained due to scale ambiguity (the problem is underconstrained). In the scenario of indoor monitoring, considered here, with people positioned on a room&#x00027;s floor, this ambiguity can be removed by assuming average height of a person (168 cm). Then, the location of a person detection (center of the bounding box) in any camera view can be mapped to 3-D room coordinates such that the 3-D point is 84 cm above the floor (vertical center of human body of average height). The inverse mapping of all person-detection locations from <italic>N</italic> camera views to 3-D space results in a point &#x0201C;cloud&#x0201D;, or rather a 2-D point &#x0201C;spatter&#x0201D; on a plane parallel to and 84 cm above the floor. We propose to <italic>cluster</italic> the locations in this &#x0201C;spatter&#x0201D; to match identities.</p>
<p>We note that, ideally, the number of location clusters should correspond to the number of people in the room. Therefore, we cannot use a clustering algorithm such as <italic>K</italic>-means (Lloyd, <xref ref-type="bibr" rid="B27">1982</xref>) since we do not know the value of <italic>K</italic> in advance. We adopt DBSCAN (Ester et al., <xref ref-type="bibr" rid="B17">1996</xref>) since it requires no advance knowledge of the number of clusters. DBSCAN is a density-based clustering algorithm that has two parameters, &#x003F5; and <italic>minPoints</italic>. In DBSCAN, first one picks a random point as the point of interest and finds all points that are within radius &#x003F5; from it. All such points, including the point of interest, get assigned to the same cluster. This process is repeated; each point in the cluster is treated as the new point of interest, thus enlarging the cluster. One continues to spread out the cluster until there is no point within &#x003F5; distance from any of the points in the cluster. Then, one picks another point from the dataset that has not been visited yet and repeats the process. For a group of points to be considered a cluster, there should be at least <italic>minPoints</italic> elements in the cluster. Also, if a certain data point has no other data points within &#x003F5; radius, it is labeled as noise and gets discarded.</p>
<p>Similarly to DBSCAN, we propose clustering of the mapped real-world coordinates using two parameters: &#x003F5; and <italic>maxPoints</italic>. While &#x003F5; has the same role as in DBSCAN, <italic>maxPoints</italic> is used differently. In DBSCAN, the size of a cluster has a lower bound of <italic>minPoints</italic> with no upper bound. In our case, cluster size must be between 1 and <italic>maxPoints</italic> due to the nature of person re-identification and people counting that we are tackling. Each person should have their own cluster, where each point in the cluster corresponds to a detection of the same person in a different camera view. In re-identification across <italic>N</italic> cameras, some people can get detected in one camera view only due to occlusions or failed detections, potentially resulting in a single point in their cluster. On the other hand, a person can be detected in at most <italic>N</italic> camera views, so a cluster may have at most <italic>N</italic> points and so <italic>maxPoints</italic> = <italic>N</italic>.</p>
<p>In experiments reported in the next section, we use 3-D Euclidean distance as the distance measure between the mapped real-world locations. However, since all such locations occur on a plane, as discussed above, effectively this is a 2-D distance in real-world coordinates.</p>
</sec>
<sec>
<title>7.3 Experimental results for 3 cameras</title>
<p>We evaluate the people-counting performance of both <italic>N</italic>-camera person ReID algorithms on the 3-day dataset captured by three cameras, that we introduced in Section 6.2. While in <xref ref-type="table" rid="T5">Table 5</xref> we reported people-counting results using only camera &#x00023;1 and in <xref ref-type="table" rid="T6">Table 6</xref> using cameras &#x00023;2 and &#x00023;3, here we report results using all three cameras installed in the classroom.</p>
<p>We use location-based re-identification with the PPD distance metric to evaluate both <italic>N</italic>-camera algorithms. Note, that in the <italic>N</italic>-D score-matrix approach elements of <bold>S</bold> (PPD values) are expressed in pixels (i.e., the lower the score/distance, the more similar the identities). While the greedy algorithm (<xref ref-type="fig" rid="F5">Figure 5</xref>) performs identity matching until no more matches are possible, some late matches may be unlikely if the corresponding element in <bold>S</bold> is large. To avoid such unlikely matches, we introduce a distance threshold &#x003BB; (in pixels), and stop the matching once all remaining elements in <bold>S</bold> exceed &#x003BB;. The <italic>N</italic>-camera location-clustering approach (Section 7.2) also has a tuning parameter &#x003F5;, expressed in centimeters, that quantifies a threshold on distance in real-world coordinates.</p>
<p>In <xref ref-type="fig" rid="F7">Figure 7</xref>, we show the people-counting performance of both <italic>N</italic>-camera algorithms (<italic>N</italic> &#x0003D; 3) when their respective tuning parameters vary. The <italic>N</italic>-D score-matrix approach yields the lowest <italic>MAE</italic> value for &#x003BB; &#x0003D; 400 pixels, while the real-world location-clustering approach achieves the best performance for &#x003F5; &#x0003D; 250 cm. We note that these values are relatively large considering the fact that we are working with 2,048 &#x000D7; 2,048-pixel images in a 22 &#x000D7; 8.5 m room. Very likely some identity matches are incorrect (as are some RAPiD detections) and yet the people count is quite accurate. However, since our dataset is labeled for people-counting only (no bounding boxes or identity labels), we cannot report ReID accuracy to verify this hypothesis.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>People-counting performance (cumulative <italic>MAE</italic> across 3 days of the test) for: <bold>(A)</bold> <italic>N</italic>-D score-matrix approach with varying &#x003BB;; and <bold>(B)</bold> real-world location-clustering approach with varying &#x003F5;, both for <italic>N</italic> &#x0003D; 3.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0007.tif"/>
</fig>
<p>While a low <italic>MAE</italic> is maintained by the <italic>N</italic>-D score-matrix approach for a wide range of &#x003BB; values (from about 100 pixels to 1,000 pixels), a high-performance range for the real-world location clustering approach happens only for &#x003F5; between about 200 and 300 cm. Outside of these ranges, <italic>MAE</italic> increases and especially rapidly for small parameter values. For example, small values of threshold &#x003F5; allow little room for image-to-3-D mapping errors; imprecisely mapped locations get absorbed into incorrect clusters. The more accurate the mapping algorithm, the smaller the value of &#x003F5; that can be used. In the extreme case of &#x003F5; &#x0003D; 0 cm, there is no room for mapping errors. Unless same-identity locations from all cameras are mapped to the same 3-D location, they cannot form one cluster. Since error-free mappings are very unlikely, for &#x003F5; &#x0003D; 0 cm very few identities can be matched, thus resulting in small <inline-formula><mml:math id="M31"><mml:msubsup><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> in <xref ref-type="disp-formula" rid="E2">Equation (1)</xref> and causing overcounting. As &#x003F5; increases, the degree of overcounting gets reduced. At the other extreme, if &#x003F5; is too large the mapped locations of different identities might fall into the same cluster thus potentially resulting in too many matches and leading to undercounting. Similar conclusions can be drawn for the impact of &#x003BB; on performance of the <italic>N</italic>-D score-matrix algorithm.</p>
<p>In order to confirm our observations about overcounting and undercounting, <xref ref-type="fig" rid="F8">Figure 8</xref> shows the true occupancy during the 3-day test (red line) and occupancy estimated by the <italic>N</italic>-D score-matrix algorithm (blue line) for three values of &#x003BB;: 100, 400 and 1,500 pixels. Notably, for &#x003BB; &#x0003D; 100 pixels the algorithm significantly overcounts, while for &#x003BB; &#x0003D; 1, 500 pixels it largely undercounts. However, for &#x003BB; &#x0003D; 400 pixels, it closely follows the true people count. <xref ref-type="fig" rid="F9">Figure 9</xref> shows similar results for the real-world location-clustering algorithm and three values of &#x003F5;: 50, 250, and 800 cm. Again, for &#x003F5; &#x0003D; 50 cm the algorithm severely overcounts, for &#x003F5; &#x0003D; 800 cm it largely undercounts, and for &#x003F5; &#x0003D; 250 cm it is most accurate.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Ground-truth people count (red line) and an estimate by 3-camera score-matrix algorithm (blue line) for three values of threshold &#x003BB;. <bold>(A)</bold> &#x003BB; = 100 pixels. <bold>(B)</bold> &#x003BB; = 400 pixels. <bold>(C)</bold> &#x003BB; = 1,500 pixels.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0008.tif"/>
</fig>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Ground-truth people count (red line) and an estimate by 3-camera real-world location-clustering algorithm (blue line) for three values of threshold <italic>&#x003F5;</italic>. <bold>(A)</bold> <italic>&#x003F5;</italic> = 50 cm. <bold>(B)</bold> <italic>&#x003F5;</italic> = 250 cm. <bold>(C)</bold> <italic>&#x003F5;</italic> = 800 cm.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimag-03-1387543-g0009.tif"/>
</fig>
<p><xref ref-type="table" rid="T7">Table 7</xref> quantifies performance of both algorithms in terms of <italic>MAE</italic>, <italic>MAE</italic><sub><italic>pp</italic></sub>, <italic>Acc</italic><sub><italic>X</italic></sub> for the tuning parameters that yielded the smallest cumulative <italic>MAE</italic> value (<xref ref-type="fig" rid="F7">Figure 7</xref>), namely &#x003BB; &#x0003D; 400 pixels and &#x003F5; &#x0003D; 250 cm. For ease of comparison, included are also results for the 2-camera people-counting algorithms from <xref ref-type="table" rid="T6">Table 6</xref> that employ greedy search of a 2-D score matrix populated by either PPD or CBD values. We note, that in terms of cumulative metrics the 3-camera score-matrix algorithm using the PPD metric slightly outperforms the same algorithm using 2 cameras. However, it performs equally well-compared to the 2-camera score-matrix algorithm with the CBD metric in terms of <italic>MAE</italic> and <italic>MAE</italic><sub><italic>pp</italic></sub>, and results are mixed in terms of <italic>Acc</italic><sub><italic>X</italic></sub> (slightly better for <italic>X</italic> &#x0003D; 5 and 10, and slightly worse for <italic>X</italic> &#x0003D; 0). As for the 3-camera real-world location-clustering algorithm, it does not perform as well; its cumulative <italic>MAE</italic> and <italic>MAE</italic><sub><italic>pp</italic></sub> values are higher by 0.07 and 0.04, respectively, than corresponding values for the 3-camera score-matrix algorithm, and the <italic>Acc</italic><sub><italic>X</italic></sub> values are lower by up to 1% point.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>People-counting performance of the RAPiD detection algorithm combined with 2-camera or 3-camera location-based re-identification algorithms in a 3-day test in the large classroom (<xref ref-type="fig" rid="F2">Figure 2C</xref>).</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Algorithm</bold></th>
<th valign="top" align="center"><italic><bold>N</bold></italic></th>
<th valign="top" align="center"><bold>Day 1</bold></th>
<th valign="top" align="center"><bold>Day 2</bold></th>
<th valign="top" align="center"><bold>Day 3</bold></th>
<th valign="top" align="center"><bold>Cumulative</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Average number of people</td>
<td/>
<td/>
<td valign="top" align="center">29.5</td>
<td valign="top" align="center">20.6</td>
<td valign="top" align="center">0.99</td>
<td valign="top" align="center">16.4</td>
</tr>
<tr>
<td valign="top" align="left"><italic>MAE&#x02193;</italic></td>
<td valign="top" align="center">2-D score matrix (PPD)</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2.42</td>
<td valign="top" align="center">1.74</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">1.63</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">2-D score matrix (CBD)</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2.43</td>
<td valign="top" align="center">1.71</td>
<td valign="top" align="center"><bold>0.75</bold></td>
<td valign="top" align="center"><bold>1.59</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">3-D score matrix (PPD)</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center"><bold>2.31</bold></td>
<td valign="top" align="center"><bold>1.62</bold></td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center"><bold>1.59</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Real-world location clustering</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">2.45</td>
<td valign="top" align="center">1.68</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">1.66</td>
</tr>
<tr>
<td valign="top" align="left"><italic>MAE</italic><sub><italic>pp</italic></sub>&#x02193;</td>
<td valign="top" align="center">2-D score matrix (PPD)</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0.082</td>
<td valign="top" align="center">0.084</td>
<td valign="top" align="center">0.849</td>
<td valign="top" align="center">0.100</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">2-D score matrix (CBD)</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0.082</td>
<td valign="top" align="center">0.083</td>
<td valign="top" align="center"><bold>0.752</bold></td>
<td valign="top" align="center"><bold>0.097</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">3-D score matrix (PPD)</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center"><bold>0.078</bold></td>
<td valign="top" align="center"><bold>0.079</bold></td>
<td valign="top" align="center">0.941</td>
<td valign="top" align="center"><bold>0.097</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Real-world location clustering</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">0.083</td>
<td valign="top" align="center">0.082</td>
<td valign="top" align="center">0.952</td>
<td valign="top" align="center">0.101</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Acc</italic><sub><italic>X</italic></sub> [%] &#x02191;<italic>X</italic>=0/5/10</td>
<td valign="top" align="center">2-D score matrix (PPD)</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center"><bold>41</bold>/84/<bold>96</bold></td>
<td valign="top" align="center">36/92/99</td>
<td valign="top" align="center">50/98/<bold>100</bold></td>
<td valign="top" align="center">42/92/98</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">2-D score matrix (CBD)</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center"><bold>41</bold>/84/<bold>96</bold></td>
<td valign="top" align="center"><bold>37</bold>/93/99</td>
<td valign="top" align="center"><bold>51</bold>/<bold>99</bold>/<bold>100</bold></td>
<td valign="top" align="center"><bold>43</bold>/92/98</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">3-D score matrix (PPD)</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">39/<bold>86</bold>/<bold>96</bold></td>
<td valign="top" align="center">35/<bold>95</bold>/<bold>100</bold></td>
<td valign="top" align="center">48/98/<bold>100</bold></td>
<td valign="top" align="center">41/<bold>93</bold>/<bold>99</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Real-world location clustering</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">39/85/95</td>
<td valign="top" align="center">35/94/99</td>
<td valign="top" align="center">49/98/<bold>100</bold></td>
<td valign="top" align="center">41/92/98</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best performance is shown in boldface.</p>
</table-wrap-foot>
</table-wrap>
<p>With respect to different occupancy scenarios, the 3-camera score-matrix approach quite consistently outperforms other algorithms on days 1 and 2 (high average occupancy), but is outperformed by the 2-camera approaches on day 3 (mostly empty). Interestingly, on day 3 the single-camera results (<xref ref-type="table" rid="T5">Table 5</xref>) are even better than those for two cameras; for example, camera &#x00023;1 mounted in the center of the classroom produces <italic>MAE</italic> and <italic>MAE</italic><sub><italic>pp</italic></sub> twice lower than those for the 2-camera algorithms. This is due to the fact that the few occupants on day 3 are located in classroom center which is effectively covered by camera &#x00023;1. With no challenges (people are directly under the camera, no occlusions, no occupants at FOV periphery), people counting using one camera is erroneous only if people detection fails. However, when multiple cameras are needed for large-area coverage, additional errors may be introduced by person ReID. Clearly, under spatially-localized low occupancy, single-camera RAPiD may be sufficient. However, in large spaces with spatially-distributed high occupancy, a multi-camera people-counting system is essential.</p>
<p>The results in <xref ref-type="table" rid="T7">Table 7</xref> may seem somewhat disappointing since the better 3-camera approach only slightly outperforms the best 2-camera approach and only in crowded scenarios. However, there is a reason for this. The 2-camera score-matrix (CBD) approach performs very well in this 187 m<sup>2</sup> test space producing cumulative <italic>MAE</italic><sub><italic>pp</italic></sub> of only 0.097 (9.7% error per person) vastly outperforming the 1-camera RAPiD performance (no re-identification) that produced a 0.373 cumulative <italic>MAE</italic><sub><italic>pp</italic></sub> (or 37.3% error per person) as shown in <xref ref-type="table" rid="T5">Table 5</xref>. As reported in Konrad et al. (<xref ref-type="bibr" rid="B21">2024</xref>), RAPiD applied to a single-camera video stream can deliver <italic>MAE</italic><sub><italic>pp</italic></sub> of 0.065 (6.5% error per person) up to about 75 m<sup>2</sup> (&#x02248;800 ft<sup>2</sup>) of a square-room area and 0.133 (or 13.3%) up to about 116 ft<sup>2</sup> (&#x02248;1,250 ft<sup>2</sup>). Considering that this is a 22 &#x000D7; 8.5 m space and that each of the cameras used by the 2-camera score-matrix approach (cameras &#x00023;2 and &#x00023;3 in <xref ref-type="fig" rid="F2">Figure 2</xref>) roughly covers one half of the classroom (about 11 &#x000D7; 8.5 m space with 93.5 m<sup>2</sup> area), it is clear that little improvement can be expected from additional cameras in this case. However, the <italic>N</italic>-camera ReID algorithms proposed in this paper are expected to be highly beneficial in larger spaces in which two cameras would be insufficient. We believe that the proposed <italic>N</italic>-camera methodology would be very valuable in scaling up people-counting to much larger spaces such as convention halls, food courts, airport terminals, train/bus stations, etc.</p>
<p>Theoretically, a fisheye camera with FOV covering 360&#x000B0; in plane parallel to the sensor and at least 180&#x000B0; orthogonally should capture an area of any size. However, due to radial distortions of its lens and finite sensor resolution, details captured at FOV periphery are insufficient for reliable person detection (and re-identification). As discussed above, in our test scenario (<xref ref-type="fig" rid="F2">Figure 2C</xref>) a single camera mounted 3.15 m above the floor can produce reliable detections up to about 8.7 &#x000D7; 8.7 m square area (&#x02248;75 m<sup>2</sup>). However, in a square area of 87 &#x000D7; 87 m, it is unlikely that 100 fisheye cameras used jointly by person ReID methods described in Sections 7.1 and 7.2 would produce reliable results using a location-based approach (PPD or CBD). As the physical distance between cameras increases, the bidirectional projection errors will grow due to errors in intrinsic and extrinsic parameters. Depending on the accuracy of these parameters, a very large area may need to be partitioned into sections, each monitored independently by a smaller group of cameras (e.g., 2 &#x000D7; 2 or 3 &#x000D7; 3). One could consider using the <italic>N</italic>-D score matrix approach with appearance features (instead of location), but a person captured by far apart cameras may appear dramatically smaller in one FOV than in the other posing very serious challenges for ReID. Again, small groups of nearby cameras would be more effective.</p>
</sec>
</sec>
<sec id="s8">
<title>8 Conclusions and future directions</title>
<p>There exists a significant demand for occupancy analytics in commercial buildings with applications ranging from security and space management to reduction of energy use. Technologies deployed today serve each of these needs individually (e.g., surveillance cameras for security, ID card access for space management, CO<sub>2</sub> sensing for energy reduction) and are not easily adaptable to other uses. While surveillance cameras could, in principle, serve all three applications, their usefulness is limited by their narrow field of view (many cameras would be needed, significantly complicating processing). Contrary to that, top-view fisheye cameras have a wide field of view and largely avoid occlusions, but few algorithms have been developed to date for the analysis of human presence and behavior using such cameras. In this paper, we reviewed some of the recent developments in this field and demonstrated that in small-to-medium size spaces (up to about 75 m<sup>2</sup>) one can very accurately detect (and count) people using a single overhead fisheye camera mounted about 3 m above the floor. However, in larger spaces several cameras are needed requiring additional processing to resolve ambiguities; for example, in counting and tracking one needs to match identities between cameras. To address this, we proposed two <italic>N</italic>-camera person re-identification algorithms and demonstrated their efficacy in large-space people counting.</p>
<p>Beyond detecting and counting people using overhead fisheye cameras, another challenge is in tracking. While we have been successful in re-identifying people between calibrated cameras based on location, reliable methods are needed for tracking people across the field of view of one camera and between cameras that do not have overlapping fields of view. This cannot be performed based on location, so advanced appearance-based methods are needed. Another unique challenge is in action recognition. A particular difficulty is the unusual viewpoint if an action is performed directly under the camera, not observed in traditional action recognition. Also, if a person moves away from under the camera while performing an action, even just a few meters, a dramatic viewpoint change occurs, again uncommon in typical action recognition studied today. There is also an application-specific challenge. In this work, people are localized in image coordinates but for security applications it would be of interest to map these locations to 2-D room layout, along the lines of real-world location clustering presented in Section 7.2. Finally, there exists a substantial performance gap between visual analysis methods developed for side-mounted rectilinear cameras and top-view fisheye cameras that needs to be closed; only then will fisheye-based indoor monitoring enter the mainstream video surveillance market. One methodology that can help achieve this goal is domain adaptation, for example by leveraging the richness of algorithms and datasets developed for front-facing fisheye cameras used in autonomous navigation. All these challenges need to be addressed before overhead fisheye cameras become ubiquitous in autonomous indoor monitoring.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s9">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/projects/vsns/cossy/datasets/">https://vip.bu.edu/projects/vsns/cossy/datasets/</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="s10">
<title>Ethics statement</title>
<p>Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec sec-type="author-contributions" id="s11">
<title>Author contributions</title>
<p>JK: Conceptualization, Funding acquisition, Methodology, Project administration, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. MC: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. MT: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. PI: Conceptualization, Funding acquisition, Methodology, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s12">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was supported by the Advanced Research Projects Agency-Energy (ARPA-E) through agreement DE-AR0000944 and by Boston University Undergraduate Research Opportunities Program (UROP).</p>
</sec>
<ack><p>The authors would like to acknowledge Boston University undergraduates Ragib Ahsan, Christopher Alonzo, John Bolognino, Annette Hong, Nancy Zheng, and Jakub Z&#x000F3;&#x00142;ko&#x0015B; for helping annotate fisheye-image datasets used in this research.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s13">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/rapid"><monospace>vip.bu.edu/rapid</monospace></ext-link></p></fn>
<fn id="fn0002"><p><sup>2</sup>ARPD (Minh et al., <xref ref-type="bibr" rid="B28">2021</xref>) was developed and evaluated prior to the introduction of the WEPDTOF dataset.</p></fn>
<fn id="fn0003"><p><sup>3</sup><ext-link ext-link-type="uri" xlink:href="https://vip.bu.edu/rapid-t"><monospace>vip.bu.edu/rapid-t</monospace></ext-link></p></fn>
<fn id="fn0004"><p><sup>4</sup>We used the 2-camera calibration method described in Section 6.4.1 except that instead of iterating through 2 camera pairs (<italic>&#x00023;K</italic>&#x02192;<italic>&#x00023;L</italic> and <italic>&#x00023;L</italic>&#x02192;<italic>&#x00023;K</italic>), we iterate through six camera pairs (<italic>&#x00023;K</italic>&#x02192;<italic>&#x00023;L, &#x00023;L</italic>&#x02192;<italic>&#x00023;K, &#x00023;K</italic>&#x02192;<italic>&#x00023;M, &#x00023;M</italic>&#x02192;<italic>&#x00023;K, &#x00023;L</italic>&#x02192;<italic>&#x00023;M, &#x00023;M</italic>&#x02192;<italic>&#x00023;L</italic>) during stochastic gradient descent.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Barman</surname> <given-names>A.</given-names></name> <name><surname>Wu</surname> <given-names>W.</given-names></name> <name><surname>Loce</surname> <given-names>R. P.</given-names></name> <name><surname>Burry</surname> <given-names>A. M.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Person re-identification using overhead view fisheye lens cameras,&#x0201D;</article-title> in <source>IEEE International Symposium on Technologies for Homeland Security (HST)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bazzani</surname> <given-names>L.</given-names></name> <name><surname>Cristani</surname> <given-names>M.</given-names></name> <name><surname>Perina</surname> <given-names>A.</given-names></name> <name><surname>Farenzena</surname> <given-names>M.</given-names></name> <name><surname>Murino</surname> <given-names>V.</given-names></name></person-group> (<year>2010</year>). <article-title>&#x0201C;Multiple-shot person re-identification by HPE signature,&#x0201D;</article-title> in <source>International Conference on Pattern Recognition</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1413</fpage>&#x02013;<lpage>1416</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blott</surname> <given-names>G.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <name><surname>Heipke</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>Multi-view person re-identification in a fisheye camera network with different viewing directions</article-title>. <source>PFG J. Photogr. Remote Sens. Geoinf. Sci</source>. <volume>87</volume>, <fpage>263</fpage>&#x02013;<lpage>274</lpage>. <pub-id pub-id-type="doi">10.1007/s41064-019-00083-y</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bone</surname> <given-names>J.</given-names></name> <name><surname>Cokbas</surname> <given-names>M.</given-names></name> <name><surname>Tezcan</surname> <given-names>O.</given-names></name> <name><surname>Konrad</surname> <given-names>J.</given-names></name> <name><surname>Ishwar</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Geometry-based person reidentification in fisheye stereo,&#x0201D;</article-title> in <source>IEEE International Conference on Advanced Video and Signal Based Surveillance (AVSS)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B5">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chiang</surname> <given-names>A.-T.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;Human detection in fish-eye images using HOG-based detectors over rotated windows,&#x0201D;</article-title> in <source>IEEE International Conference on Multimedia and Expo Workshops (ICMEW)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chiang</surname> <given-names>S.-H.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Chen</surname> <given-names>Y.-F.</given-names></name></person-group> (<year>2021</year>). <article-title>Efficient pedestrian detection in top-view fisheye images using compositions of perspective view patches</article-title>. <source>Image Vis. Comput</source>. <volume>105</volume>:<fpage>104069</fpage>. <pub-id pub-id-type="doi">10.1016/j.imavis.2020.104069</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cokbas</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <source>Person Re-identification Using Fisheye Cameras With Application to Occupancy Analysis</source> (<publisher-loc>PhD thesis</publisher-loc>). Boston University, Boston, MA, United States.</citation>
</ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cokbas</surname> <given-names>M.</given-names></name> <name><surname>Bolognino</surname> <given-names>J.</given-names></name> <name><surname>Konrad</surname> <given-names>J.</given-names></name> <name><surname>Ishwar</surname> <given-names>P.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;FRIDA: fisheye re-identification dataset with annotations,&#x0201D;</article-title> in <source>IEEE International Conference on Advanced Video and Signal Based Surveillance (AVSS)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cokbas</surname> <given-names>M.</given-names></name> <name><surname>Ishwar</surname> <given-names>P.</given-names></name> <name><surname>Konrad</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Spatio-visual fusion-based person re-identification for overhead fisheye images</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>46095</fpage>&#x02013;<lpage>46106</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3274600</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cordts</surname> <given-names>M.</given-names></name> <name><surname>Omran</surname> <given-names>M.</given-names></name> <name><surname>Ramos</surname> <given-names>S.</given-names></name> <name><surname>Rehfeld</surname> <given-names>T.</given-names></name> <name><surname>Enzweiler</surname> <given-names>M.</given-names></name> <name><surname>Benenson</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;The Cityscapes dataset for semantic urban scene understanding,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3213</fpage>&#x02013;<lpage>3223</lpage>.<pub-id pub-id-type="pmid">32191886</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Courbon</surname> <given-names>J.</given-names></name> <name><surname>Mezouar</surname> <given-names>Y.</given-names></name> <name><surname>Martinet</surname> <given-names>P.</given-names></name></person-group> (<year>2012</year>). <article-title>Evaluation of the unified model of the sphere for fisheye cameras in robotic applications</article-title>. <source>Adv. Robot</source>. <volume>26</volume>, <fpage>947</fpage>&#x02013;<lpage>967</lpage>. <pub-id pub-id-type="doi">10.1163/156855312X633057</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dalal</surname> <given-names>N.</given-names></name> <name><surname>Triggs</surname> <given-names>B.</given-names></name></person-group> (<year>2005</year>). <article-title>&#x0201C;Histograms of oriented gradients for human detection,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>886</fpage>&#x02013;<lpage>893</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Demirkus</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Eschey</surname> <given-names>M.</given-names></name> <name><surname>Kaestle</surname> <given-names>H.</given-names></name> <name><surname>Galasso</surname> <given-names>F.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;People detection in fish-eye top-views,&#x0201D;</article-title> in <source>International Joint Conference on Computer Vision, Imaging and Computer Graphics Theory and Applications (VISIGRAPP 2017), Vol. 5</source> (<publisher-loc>SciTePress</publisher-loc>), <fpage>141</fpage>&#x02013;<lpage>148</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name> <name><surname>Appel</surname> <given-names>R.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name> <name><surname>Perona</surname> <given-names>P.</given-names></name></person-group> (<year>2014</year>). <article-title>Fast feature pyramids for object detection</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>36</volume>, <fpage>1532</fpage>&#x02013;<lpage>1545</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2014.2300479</pub-id><pub-id pub-id-type="pmid">26353336</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name> <name><surname>Fischer</surname> <given-names>P.</given-names></name> <name><surname>Ilg</surname> <given-names>E.</given-names></name> <name><surname>H&#x000E4;usser</surname> <given-names>P.</given-names></name> <name><surname>Hazirbas</surname> <given-names>C.</given-names></name> <name><surname>Golkov</surname> <given-names>V.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>&#x0201C;FlowNet: learning optical flow with convolutional networks,&#x0201D;</article-title> in <source>IEEE International Conference on Computer Vision</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2758</fpage>&#x02013;<lpage>2766</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>Z.</given-names></name> <name><surname>Ozan</surname> <given-names>T. M.</given-names></name> <name><surname>Nakamura</surname> <given-names>H.</given-names></name> <name><surname>Ishwar</surname> <given-names>P.</given-names></name> <name><surname>Konrad</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;RAPiD: rotation-aware people detection in overhead fisheye images,&#x0201D;</article-title> in <source>IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ester</surname> <given-names>M.</given-names></name> <name><surname>Kriegel</surname> <given-names>H.-P.</given-names></name> <name><surname>Sander</surname> <given-names>J.</given-names></name> <name><surname>Xu</surname> <given-names>X.</given-names></name></person-group> (<year>1996</year>). <article-title>&#x0201C;A density-based algorithm for discovering clusters in large spatial databases with noise,&#x0201D;</article-title> in <source>International Conference on Knowledge Discovery and Data Mining</source> (<publisher-loc>AAAI Press</publisher-loc>), <fpage>226</fpage>&#x02013;<lpage>231</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Farneb&#x000E4;ck</surname> <given-names>G.</given-names></name></person-group> (<year>2003</year>). <article-title>&#x0201C;Two-frame motion estimation based on polynomial expansion,&#x0201D;</article-title> in <source>Image Analysis</source>, eds. J. Bigun, and T. Gustavsson (Berlin, Heidelberg: Springer Berlin Heidelberg), <fpage>363</fpage>&#x02013;<lpage>370</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geyer</surname> <given-names>C.</given-names></name> <name><surname>Danilidis</surname> <given-names>K.</given-names></name></person-group> (<year>2001</year>). <article-title>Catadioptric projective geometry</article-title>. <source>Int. J. Comp. Vision</source> <volume>45</volume>, <fpage>223</fpage>&#x02013;<lpage>243</lpage>. <pub-id pub-id-type="doi">10.1023/A:1013610201135</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Fast R-CNN,&#x0201D;</article-title> in <source>IEEE International Conference on Computer Vision</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1440</fpage>&#x02013;<lpage>1448</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Konrad</surname> <given-names>J.</given-names></name> <name><surname>Cokbas</surname> <given-names>M.</given-names></name> <name><surname>Ishwar</surname> <given-names>P.</given-names></name> <name><surname>Little</surname> <given-names>T. D.</given-names></name> <name><surname>Gevelber</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>High-accuracy people counting in large spaces using overhead fisheye cameras</article-title>. <source>Energy Build</source>. <volume>307</volume>:<fpage>113936</fpage>. <pub-id pub-id-type="doi">10.1016/j.enbuild.2024.113936</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Krams</surname> <given-names>O.</given-names></name> <name><surname>Kiryati</surname> <given-names>N.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;People detection in top-view fisheye imaging,&#x0201D;</article-title> in <source>IEEE International Conference on Advanced Video and Signal-Based Surveillance (AVSS)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).<pub-id pub-id-type="pmid">35448242</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Tezcan</surname> <given-names>M. O.</given-names></name> <name><surname>Ishwar</surname> <given-names>P.</given-names></name> <name><surname>Konrad</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Supervised people counting using an overhead fisheye camera,&#x0201D;</article-title> in <source>IEEE International Conference on Advanced Video and Signal-Based Surveillance (AVSS)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>Y.</given-names></name> <name><surname>Xie</surname> <given-names>J.</given-names></name> <name><surname>Geiger</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>KITTI-360: a novel dataset and benchmarks for urban scene understanding in 2D and 3D</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>45</volume>, <fpage>3292</fpage>&#x02013;<lpage>3310</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2022.3179507</pub-id><pub-id pub-id-type="pmid">35648872</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Maire</surname> <given-names>M.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name> <name><surname>Hays</surname> <given-names>J.</given-names></name> <name><surname>Perona</surname> <given-names>P.</given-names></name> <name><surname>Ramanan</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>&#x0201C;Microsoft COCO: common objects in context,&#x0201D;</article-title> in <source>Computer Vision-ECCV 2014</source>, eds. D. Fleet, T. Pajdla, B. Schiele, and T. Tuytelaars (Cham: Springer International Publishing), <fpage>740</fpage>&#x02013;<lpage>755</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Anguelov</surname> <given-names>D.</given-names></name> <name><surname>Erhan</surname> <given-names>D.</given-names></name> <name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Reed</surname> <given-names>S.</given-names></name> <name><surname>Fu</surname> <given-names>C.-Y.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;SSD: single shot multibox detector,&#x0201D;</article-title> in <source>Computer Vision-ECCV 2016</source>, eds. B. Leibe, J. Matas, N. Sebe, and M. Welling (Cham: Springer International Publishing), <fpage>21</fpage>&#x02013;<lpage>37</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lloyd</surname> <given-names>S.</given-names></name></person-group> (<year>1982</year>). <article-title>Least squares quantization in PCM</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>28</volume>, <fpage>129</fpage>&#x02013;<lpage>137</lpage>. <pub-id pub-id-type="doi">10.1109/TIT.1982.1056489</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Minh</surname> <given-names>Q. N.</given-names></name> <name><surname>Van</surname> <given-names>B. L.</given-names></name> <name><surname>Nguyen</surname> <given-names>C.</given-names></name> <name><surname>Le</surname> <given-names>A.</given-names></name> <name><surname>Nguyen</surname> <given-names>V. D.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;ARPD: anchor-free rotation-aware people detection using topview fisheye camera,&#x0201D;</article-title> in <source>IEEE International Conference on Advanced Video and Signal-Based Surveillance (AVSS)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B29">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Panda</surname> <given-names>R.</given-names></name> <name><surname>Bhuiyan</surname> <given-names>A.</given-names></name> <name><surname>Murino</surname> <given-names>V.</given-names></name> <name><surname>Roy-Chowdhury</surname> <given-names>A. K.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Unsupervised adaptive re-identification in open world dynamic camera networks,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1377</fpage>&#x02013;<lpage>1386</lpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Redmon</surname> <given-names>J.</given-names></name> <name><surname>Divvala</surname> <given-names>S.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Farhadi</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;You only look once: unified, real-time object detection,&#x0201D;</article-title> in <source>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>779</fpage>&#x02013;<lpage>788</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Redmon</surname> <given-names>J.</given-names></name> <name><surname>Farhadi</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <source>Yolov3: An Incremental Improvement. CoRR, abs/1804.02767</source>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Faster R-CNN: towards real-time object detection with region proposal networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, Vol. 28</source>, eds. C. Cortes, N. Lawrence, D. Lee, M. Sugiyama, and R. Garnett (Curran Associates, Inc.).</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sabater</surname> <given-names>A.</given-names></name> <name><surname>Montesano</surname> <given-names>L.</given-names></name> <name><surname>Murillo</surname> <given-names>A. C.</given-names></name></person-group> (<year>2020</year>). <article-title>Robust and efficient post-processing for video object detection</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.1109/IROS45743.2020.9341600</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Saito</surname> <given-names>M.</given-names></name> <name><surname>Kitaguchi</surname> <given-names>K.</given-names></name> <name><surname>Kimura</surname> <given-names>G.</given-names></name> <name><surname>Hashimoto</surname> <given-names>M.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;People detection and tracking from fish-eye image based on probabilistic appearance model,&#x0201D;</article-title> in <source>SICE Annual Conference</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>435</fpage>&#x02013;<lpage>440</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seidel</surname> <given-names>R.</given-names></name> <name><surname>Apitzsch</surname> <given-names>A.</given-names></name> <name><surname>Hirtz</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>Improved person detection on omnidirectional images with non-maxima suppression</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.5220/0007388400002108</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tamura</surname> <given-names>M.</given-names></name> <name><surname>Horiguchi</surname> <given-names>S.</given-names></name> <name><surname>Murakami</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Omnidirectional pedestrian detection by rotation invariant training,&#x0201D;</article-title> in <source>IEEE Winter Conference on Applications of Computer Vision (WACV)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tamura</surname> <given-names>M.</given-names></name> <name><surname>Yoshinaga</surname> <given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>Segmentation-based bounding box generation for omnidirectional pedestrian detection</article-title>. <source>Visual Comp</source>. <volume>40</volume>, <fpage>2505</fpage>&#x02013;<lpage>2516</lpage>. <pub-id pub-id-type="doi">10.1007/s00371-023-02933-8</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tezcan</surname> <given-names>M. O.</given-names></name> <name><surname>Duan</surname> <given-names>Z.</given-names></name> <name><surname>Cokbas</surname> <given-names>M.</given-names></name> <name><surname>Ishwar</surname> <given-names>P.</given-names></name> <name><surname>Konrad</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;WEPDTOF: a dataset and benchmark algorithms for in-the-wild people detection and tracking from overhead fisheye cameras,&#x0201D;</article-title> in <source>IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1381</fpage>&#x02013;<lpage>1390</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>van Lint</surname> <given-names>J. H.</given-names></name> <name><surname>Wilson</surname> <given-names>R. M.</given-names></name></person-group> (<year>1992</year>). <source>Chapter 10: A Course in Combinatorics</source> (<publisher-loc>Cambridge University Press</publisher-loc>).</citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Chiang</surname> <given-names>S.-H.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Online pedestrian tracking using a dense fisheye camera network with edge computing,&#x0201D;</article-title> in <source>IEEE International Conference on Image Processing (ICIP)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3518</fpage>&#x02013;<lpage>3522</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>X.</given-names></name> <name><surname>Wei</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>RMDC: rotation-mask deformable convolution for object detection in top-view fisheye cameras</article-title>. <source>Neurocomputing</source> <volume>504</volume>, <fpage>99</fpage>&#x02013;<lpage>108</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2022.06.116</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ye</surname> <given-names>M.</given-names></name> <name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Lin</surname> <given-names>G.</given-names></name> <name><surname>Xiang</surname> <given-names>T.</given-names></name> <name><surname>Shao</surname> <given-names>L.</given-names></name> <name><surname>Hoi</surname> <given-names>S. H.</given-names></name></person-group> (<year>2022</year>). <article-title>Deep learning for person re-identification: a survey and outlook</article-title>. <source>IEEE Trans. Pattern Anal. Machine Intell</source>. <volume>44</volume>, <fpage>2872</fpage>&#x02013;<lpage>2893</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2021.3054775</pub-id><pub-id pub-id-type="pmid">33497329</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ye</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>K.</given-names></name> <name><surname>Xiang</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>K.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Universal semantic segmentation for fisheye urban driving images,&#x0201D;</article-title> in <source>IEEE International Conference on Systems, Man, and Cybernetics (SMC)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>648</fpage>&#x02013;<lpage>655</lpage>.<pub-id pub-id-type="pmid">30691055</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yogamani</surname> <given-names>S.</given-names></name> <name><surname>Hughes</surname> <given-names>C.</given-names></name> <name><surname>Horgan</surname> <given-names>J.</given-names></name> <name><surname>Sistu</surname> <given-names>G.</given-names></name> <name><surname>Chennupati</surname> <given-names>S.</given-names></name> <name><surname>Uricar</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Woodscape: a multi-task, multi-camera fisheye dataset for autonomous driving,&#x0201D;</article-title> in <source>IEEE International Conference on Computer Vision</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>9307</fpage>&#x02013;<lpage>9317</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>J.</given-names></name> <name><surname>Grassi</surname> <given-names>A. C. P.</given-names></name> <name><surname>Hirtz</surname> <given-names>G.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Applications of deep learning for top-view omnidirectional imaging: a survey,&#x0201D;</article-title> in <source>IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>P.</given-names></name> <name><surname>Yu</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Zheng</surname> <given-names>J.</given-names></name> <name><surname>Ning</surname> <given-names>X.</given-names></name> <name><surname>Bai</surname> <given-names>X.</given-names></name></person-group> (<year>2024</year>). <article-title>Towards effective person search with deep learning: a survey from systematic perspective</article-title>. <source>Pattern Recognit</source>. <volume>152</volume>:<fpage>110434</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2024.110434</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Dai</surname> <given-names>J.</given-names></name> <name><surname>Yuan</surname> <given-names>L.</given-names></name> <name><surname>Wei</surname> <given-names>Y.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Flow-guided feature aggregation for video object detection,&#x0201D;</article-title> in <source>IEEE International Conference on Computer Vision</source> (<publisher-loc>Piscataway, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>408</fpage>&#x02013;<lpage>417</lpage>.<pub-id pub-id-type="pmid">37027778</pub-id></citation></ref>
</ref-list>
</back>
</article>