<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" 'JATS-journalpublishing1-3-mathml3.dtd'>
<article article-type="review-article" dtd-version="1.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title-group>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1658643</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1658643</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Review</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Social robot navigation: a review and benchmarking of learning-based methods</article-title>
<alt-title alt-title-type="left-running-head">Alyassi et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1658643">10.3389/frobt.2025.1658643</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Alyassi</surname>
<given-names>Rashid</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3099737"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cadena</surname>
<given-names>Cesar</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Riener</surname>
<given-names>Robert</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Paez-Granados</surname>
<given-names>Diego</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>Spinal Cord Injury and Artificial Intelligence Lab, D-HEST, ETH Zurich</institution>, <city>Z&#xfc;rich</city>, <country country="CH">Switzerland</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>Sensory-Motor Systems Lab, Institute of Robotics and Intelligent Systems, ETH Zurich</institution>, <city>Z&#xfc;rich</city>, <country country="CH">Switzerland</country>
</aff>
<aff id="aff3">
<label>3</label>
<institution>Digital Healthcare and Rehabilitation, Swiss Paraplegic Research</institution>, <city>Nottwil</city>, <country country="CH">Switzerland</country>
</aff>
<aff id="aff4">
<label>4</label>
<institution>Robotics Systems Lab, Institute of Robotics and Intelligent Systems, ETH Zurich</institution>, <city>Z&#xfc;rich</city>, <country country="CH">Switzerland</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Rashid Alyassi, <email xlink:href="mailto:ralyassi@ethz.ch">ralyassi@ethz.ch</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-12-11">
<day>11</day>
<month>12</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1658643</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>15</day>
<month>10</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>27</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Alyassi, Cadena, Riener and Paez-Granados.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Alyassi, Cadena, Riener and Paez-Granados</copyright-holder>
<license>
<ali:license_ref start_date="2025-12-11">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>For autonomous mobile robots to operate effectively in human environments, navigation must extend beyond obstacle avoidance to incorporate social awareness. Safe and fluid interaction in shared spaces requires the ability to interpret human motion and adapt to social norms&#x2014;an area that is being reshaped by advances in learning-based methods. This review examines recent progress in learning-based social navigation methods that deal with the complexities of human-robot coexistence. We introduce a taxonomy of navigation methods and analyze core system components, including realistic training environments and objectives that promote socially compliant behavior. We conduct a comprehensive benchmark of existing frameworks in challenging crowd scenarios, showing their advantages and shortcomings, while providing critical insights into the architectural choices that impact performance. We find that many learning-based approaches outperform model-based methods in realistic coordination scenarios such as navigating doorways. A key highlight is the end-to-end models, which achieve strong performance by directly planning from raw sensor input, enabling more efficient and adaptive navigation. This review also maps current trends and outlines ongoing challenges, offering a strategic roadmap for future research. We emphasize the need for models that accurately anticipate human movement, training environments that realistically simulate crowded spaces, and evaluation methods that capture real-world complexity. Advancing these areas will help overcome current limitations and move social navigation systems closer to safe, reliable deployment in everyday environments. Additional resources are available at: <ext-link ext-link-type="uri" xlink:href="https://socialnavigation.github.io">https://socialnavigation.github.io</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>social navigation</kwd>
<kwd>human-robot interaction</kwd>
<kwd>reinforcement learning</kwd>
<kwd>robot learning</kwd>
<kwd>human-aware navigation</kwd>
<kwd>path planning</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This research work was partially supported by the Innosuisse Project 103.421 IP-IC &#x201c;Developing an AI-enabled Robotic Personal Vehicle for Reduced Mobility Population in Complex Environments&#x201d;.</funding-statement>
</funding-group>
<counts>
<fig-count count="5"/>
<table-count count="9"/>
<equation-count count="1"/>
<ref-count count="363"/>
<page-count count="34"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Human-Robot Interaction</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>Social navigation enables robots to move safely and efficiently in human-shared environments while respecting social norms and prioritizing human comfort. It builds on standard collision avoidance navigation by incorporating behaviors such as maintaining social distance, interpreting social cues, and predicting human movements. As a key component of <italic>Human-Robot Interaction (HRI)</italic>, social navigation focuses on understanding and enhancing interactions between humans and robots in shared environments.</p>
<p>The importance of social navigation was recognized as early as the 1990s with pioneering robots like <italic>RHINO</italic> (<xref ref-type="bibr" rid="B29">Burgard et al., 1999</xref>) and <italic>MINERVA</italic> (<xref ref-type="bibr" rid="B295">Thrun et al., 2000</xref>), which operated in dynamic environments such as museums, requiring socially aware navigation systems to interact effectively with visitors. Since then, social navigation has gained research interest, leading to steady advancements over the past years.</p>
<p>Several review papers reflect the interdisciplinary nature of social navigation. Sociological and human factors are addressed by <xref ref-type="bibr" rid="B255">Rios-Martinez et al. (2015)</xref>, who apply proxemics theory, and <xref ref-type="bibr" rid="B294">Thomaz et al. (2016)</xref>, who review computational human-robot interaction. Perception and mapping in social contexts are discussed by <xref ref-type="bibr" rid="B39">Charalampous et al. (2017)</xref>, while safety in human-robot interaction is analyzed by <xref ref-type="bibr" rid="B152">Lasota et al. (2017)</xref>. Path planning and navigation are extensively reviewed by <xref ref-type="bibr" rid="B206">Mohanan and Salgoankar (2018)</xref>, <xref ref-type="bibr" rid="B267">S&#xe1;nchez-Ib&#xe1;&#xf1;ez et al. (2021)</xref>, and <xref ref-type="bibr" rid="B356">Zhou et al. (2022)</xref>, although mainly for classical methods. For social navigation specifically, recent surveys cover human-aware navigation (<xref ref-type="bibr" rid="B150">Kruse et al., 2013</xref>), conflict prevention (<xref ref-type="bibr" rid="B203">Mirsky et al., 2021</xref>), visual navigation (<xref ref-type="bibr" rid="B207">M&#xf6;ller et al., 2021</xref>), evaluation (<xref ref-type="bibr" rid="B98">Gao and Huang, 2022</xref>; <xref ref-type="bibr" rid="B197">Mavrogiannis et al., 2023</xref>), and taxonomy (<xref ref-type="bibr" rid="B278">Singamaneni et al., 2024</xref>). Human motion prediction surveys include <xref ref-type="bibr" rid="B262">Rudenko et al. (2020a)</xref>, <xref ref-type="bibr" rid="B275">Sighencea et al. (2021)</xref>, and <xref ref-type="bibr" rid="B148">Korbmacher and Tordeux (2022)</xref>, comparing data-driven and model-based approaches. However, there remains a gap for a comprehensive survey focused on learning-based social navigation approaches.</p>
<p>This survey advances learning-based social navigation by comprehensively reviewing recent methods and introducing a novel taxonomy that categorizes algorithms into five groups by neural network architecture and system modules, expanding on earlier works like <xref ref-type="bibr" rid="B359">Zhu and Zhang (2021)</xref>. We examine key system components, including human detection, tracking, prediction, and crowd simulation. Furthermore, our conclusions are grounded in an experimental benchmark over state-of-the-art social navigation algorithms, featuring challenging scenarios such as corridors, doorways, and intersections&#x2014;areas often overlooked in previous surveys (<xref ref-type="bibr" rid="B197">Mavrogiannis et al., 2023</xref>). By rigorously comparing existing methods, we identify best practices, evaluate algorithm performance on new scenarios, and highlight open challenges and future directions, providing a comprehensive guide for developing learning-based social navigation systems.</p>
<p>The structure of this survey is as follows: <xref ref-type="sec" rid="s1">Section 1</xref> introduces a taxonomy of the social navigation problem. <xref ref-type="sec" rid="s2">Section 2</xref> presents the proposed taxonomy of social navigation algorithms and reviews recent learning-based methods. In <xref ref-type="sec" rid="s3">Section 3</xref>, we examine training processes for navigation models, including discussions on objective functions, crowd simulation, and methods for human detection, tracking, and prediction. <xref ref-type="sec" rid="s4">Section 4</xref> presents an experimental comparison to validate our analysis by evaluating multiple algorithms across various simulated scenarios. Finally, <xref ref-type="sec" rid="s5">Section 5</xref> provides a discussion of existing challenges and proposes future research directions to advance social navigation.</p>
<sec id="s1-1">
<label>1.1</label>
<title>Social navigation problem</title>
<p>Social navigation refers to a robot&#x2019;s ability to navigate environments while considering human presence, social norms, and behaviors. This field encompasses a variety of navigation tasks, broadly classified into three main categories: independent, assistive, and collaborative navigation (<xref ref-type="bibr" rid="B278">Singamaneni et al., 2024</xref>).</p>
<sec id="s1-1-1">
<label>1.1.1</label>
<title>Independent</title>
<p>Independent crowd-aware navigation involves robots autonomously reaching goals in human-populated environments while minimizing disruption, as seen with service robots in malls or airports integrating into pedestrian flows (<xref ref-type="bibr" rid="B341">Yao et al., 2019</xref>). This includes systems designed for joining moving groups (<xref ref-type="bibr" rid="B298">Truong and Ngo, 2017</xref>) or avoiding stationary crowds (<xref ref-type="bibr" rid="B302">Tsoi et al., 2022</xref>). Independent navigation is the most widely studied and versatile form of social navigation.</p>
</sec>
<sec id="s1-1-2">
<label>1.1.2</label>
<title>Assistive</title>
<p>Assistive navigation tasks involve robots directly supporting humans, such as follower robots in airports (<xref ref-type="bibr" rid="B109">Gupta et al., 2016</xref>), shopping assistants (<xref ref-type="bibr" rid="B41">Chen Y. et al., 2017</xref>), interactive guides (<xref ref-type="bibr" rid="B29">Burgard et al., 1999</xref>; <xref ref-type="bibr" rid="B295">Thrun et al., 2000</xref>), and systems aiding visually impaired individuals (<xref ref-type="bibr" rid="B58">Chuang et al., 2018</xref>), or accompanying people and groups (<xref ref-type="bibr" rid="B90">Ferrer et al., 2017</xref>; <xref ref-type="bibr" rid="B252">Repiso et al., 2020</xref>). Some tasks include proactively offering guidance (<xref ref-type="bibr" rid="B139">Kato et al., 2015</xref>). These tasks require detecting, following, and interpreting human cues for safe and seamless assistance.</p>
</sec>
<sec id="s1-1-3">
<label>1.1.3</label>
<title>Collaborative</title>
<p>Collaborative navigation features robots and humans working together on shared tasks, either physically or through shared control. In industry, cobots assist on assembly lines (<xref ref-type="bibr" rid="B193">Matheson et al., 2019</xref>), while human mobility robots use shared-control systems, model-based (<xref ref-type="bibr" rid="B105">Gonon et al., 2021</xref>) or learning-based (<xref ref-type="bibr" rid="B352">Zhang et al., 2023</xref>) to integrate human input and dynamically adapt to real-time feedback.</p>
<p>In addition to task-based classification, social navigation can be categorized by communication strategies, focusing on how robots interact with humans through signals. For a more in-depth discussion on taxonomy, refer to <xref ref-type="bibr" rid="B278">Singamaneni et al. (2024)</xref> and <xref ref-type="bibr" rid="B203">Mirsky et al. (2021)</xref>.</p>
<p>This review focuses on independent (crowd-aware) navigation due to its broad applicability. Its core principles can be extended to assistive and collaborative tasks, making it a more general foundation for various social navigation tasks.</p>
</sec>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Social navigation algorithms</title>
<p>This section explores a range of learning-based social navigation algorithms designed for crowd-aware robot navigation. These methods function as local planners and require integration with a global planner for long-term navigation. Learning-based social navigation enables robots to navigate safely around humans through trial and error or imitation. The algorithms are categorized based on their neural network architecture and the specific modules they require, such as human detection, tracking, and prediction. This classification organizes social navigation strategies into five main categories, ranging from simpler end-to-end models to sophisticated multi-policy and prediction-based methods (see <xref ref-type="fig" rid="F1">Figure 1</xref>). Furthermore, within each category, we outline several subtopics that describe common methodological themes. These themes are prevalent in certain categories but are not necessarily unique to them.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Taxonomy of Social Navigation Based on Architecture and Components outlined in <xref ref-type="sec" rid="s2-1">Sections 2.1</xref>&#x2013;<xref ref-type="sec" rid="s2-1">2.5</xref>: <bold>(a)</bold> End-to-End, <bold>(b)</bold> Human Position-based, <bold>(c)</bold> Human Attention-based, <bold>(d)</bold> Human Prediction-based, <bold>(e)</bold> Safety-aware.</p>
</caption>
<graphic xlink:href="frobt-12-1658643-g001.tif">
<alt-text content-type="machine-generated">Diagram illustrating five categories of social navigation based on their architectural components. Each row shows a sequence from a robot icon on the left to a goal on the right, with labeled processing modules in between. (a) End-to-End: the robot directly performs social navigation. (b) Human Position-based: the robot uses human tracking before social navigation. (c) Human Attention-based: the robot performs human tracking followed by human attention before navigation. (d) Human Prediction-based: the robot performs human tracking followed by human prediction before navigation. (e) Safety-aware: the robot applies safety mechanisms before and after the social navigation module.</alt-text>
</graphic>
</fig>
<sec id="s2-1">
<label>2.1</label>
<title>End-to-end navigation</title>
<p>End-to-end reinforcement learning (RL) (see <xref ref-type="table" rid="T1">Table 1</xref>) has proven highly effective across domains like robot navigation and autonomous driving (<xref ref-type="bibr" rid="B24">Bojarski, 2016</xref>). In end-to-end RL, the policy maps observations directly to actions, bypassing predefined intermediary steps and enabling complex behavior learning through trial and error. Typically, the robot&#x2019;s state <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents sensory inputs, and other navigation-related parameters such as velocity <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and goal-relative distance <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>End-to-end social navigation algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Input</th>
<th align="left">Architecture</th>
<th align="left">Algorithm</th>
<th align="left">Output</th>
<th align="left">Training sim</th>
<th align="left">Real-world demo</th>
<th align="left">Code</th>
<th align="left">Training type</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B312">Wang et al. (2018a)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">1D CNN &#x2b; FC</td>
<td align="left">Q-learning</td>
<td align="left">Discrete <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">None</td>
<td align="left">No</td>
<td align="left">Constant velocity (overlapping) obstacles</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B124">Hoeller et al. (2021)</xref>
</td>
<td align="left">RGB-D</td>
<td align="left">VAE &#x2b; LSTM &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Isaac Gym</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Constant velocity obstacles</td>
</tr>
<tr>
<td align="left">CrowdMove (<xref ref-type="bibr" rid="B83">Fan et al., 2018</xref>)</td>
<td align="left">2D LiDAR</td>
<td align="left">1D CNN &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Stage simulator</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Multi-agent</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B162">Liang et al. (2021)</xref>
</td>
<td align="left">RGB-D &#x2b; 2D LiDAR</td>
<td align="left">2D CNN &#x2b; 1D CNN &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Gazebo</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Agent-Based Crowd Sim (<xref ref-type="bibr" rid="B213">Narang et al., 2015</xref>)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B135">Jin et al. (2020)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">2D CNN &#x2b; FC</td>
<td align="left">DDPG</td>
<td align="left">
<inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">ORCA (non-cooperative) crowd</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B290">Tai et al. (2018)</xref>
</td>
<td align="left">Depth image</td>
<td align="left">ResNet-50 &#x2b;FC &#x2b; 2D CNN</td>
<td align="left">GAIL &#x2b; TRPO</td>
<td align="left">
<inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Gazebo</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">SFM crowd sim</td>
</tr>
<tr>
<td align="left">NavRep (<xref ref-type="bibr" rid="B73">Dugas et al., 2021</xref>)</td>
<td align="left">2D LiDAR</td>
<td align="left">LSTM &#x2b; VAE &#x2b; FC</td>
<td align="left">World Models &#x2b; PPO</td>
<td align="left">
<inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">ORCA &#x2b; Constant Velocity &#x2b; Global planning crowd</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B61">Cui et al. (2021)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">LSTM &#x2b; 3D CNN &#x2b; FC</td>
<td align="left">TD3</td>
<td align="left">
<inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Gazebo</td>
<td align="left">Demo</td>
<td align="left">Yes</td>
<td align="left">Multi-agent</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B117">Han et al. (2022a)</xref>
</td>
<td align="left">RGB &#x2b; 2D LiDAR</td>
<td align="left">2D CNN &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Stage simulator</td>
<td align="left">Demo</td>
<td align="left">No</td>
<td align="left">Multi-agent</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B55">Choi et al. (2019)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">2D CNN &#x2b; 1D CNN &#x2b; LSTM &#x2b; FC</td>
<td align="left">SAC</td>
<td align="left">
<inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Multi-agent</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B52">Cheng et al. (2023)</xref>
</td>
<td align="left">2D LiDAR &#x2b; Preference</td>
<td align="left">FC</td>
<td align="left">Q-Learning</td>
<td align="left">Discrete <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">None</td>
<td align="left">No</td>
<td align="left">SFM crowd sim</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B56">Choi et al. (2020)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">GRU &#x2b; FC</td>
<td align="left">SAC</td>
<td align="left">
<inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Unity Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Multi-agent</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B154">Lee et al. (2023)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">1D CNN &#x2b; FC</td>
<td align="left">SAC &#x2b; HER</td>
<td align="left">
<inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">Demo</td>
<td align="left">Yes</td>
<td align="left">Learning-based (<xref ref-type="bibr" rid="B176">Long et al., 2018</xref>) &#x2b; global planning crowd sim</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Q-learning is one of the earliest learning-based navigation methods, initially designed for static environments (<xref ref-type="bibr" rid="B279">Smart and Kaelbling, 2000</xref>; <xref ref-type="bibr" rid="B280">Smart and Kaelbling, 2002</xref>; <xref ref-type="bibr" rid="B336">Yang et al., 2004</xref>) and later extended to dynamic settings (<xref ref-type="bibr" rid="B343">Yen and Hickey, 2004</xref>; <xref ref-type="bibr" rid="B60">Costa and Gouvea, 2010</xref>; <xref ref-type="bibr" rid="B133">Jaradat et al., 2011</xref>). For instance, Wang et al. (<xref ref-type="bibr" rid="B312">Wang Y. et al., 2018</xref>) use a two-stream Q-network (<xref ref-type="bibr" rid="B276">Simonyan and Zisserman, 2014</xref>) that processes spatial (current LiDAR) and temporal (scan-difference) inputs to explicitly capture obstacle motion. These streams are processed and combined via fully connected layers, enabling effective detection of moving obstacles. While historically notable, end-to-end Q-learning is now rarely used in social navigation due to its difficulty in handling the continuous action spaces needed for smooth, realistic motion.</p>
<p>Actor-critic methods are widely used for continuous action spaces, addressing Q-learning&#x2019;s limitations. Actor-critic models have been applied to both static (<xref ref-type="bibr" rid="B289">Tai et al., 2017</xref>; <xref ref-type="bibr" rid="B349">Zhang et al., 2017</xref>; <xref ref-type="bibr" rid="B99">Gao et al., 2020</xref>) and dynamic environments (<xref ref-type="bibr" rid="B86">Faust et al., 2018</xref>; <xref ref-type="bibr" rid="B53">Chiang et al., 2019</xref>). For instance, <xref ref-type="bibr" rid="B124">Hoeller et al. (2021)</xref> employ the PPO algorithm in combination with an LSTM network to train a robot to navigate a simulated environment. To train for dynamic collision avoidance, the environment is populated with both static and dynamic (constant-velocity) obstacles.</p>
<p>An alternative to using dynamic obstacles for collision avoidance training is multi-agent reinforcement learning (MARL). MARL often leverages the concept of <italic>centralized learning with decentralized execution</italic> to develop cooperative navigation policies (<xref ref-type="bibr" rid="B350">Zhang et al., 2021</xref>). In this setup, all agents are trained within a shared environment, with each agent aiming to reach its designated goal while avoiding collisions with others (<xref ref-type="bibr" rid="B46">Chen W. et al., 2019</xref>; <xref ref-type="bibr" rid="B292">Tan et al., 2020</xref>). It&#x2019;s <italic>decentralized</italic> since there is no direct communication between agents; however, the training is <italic>centralized</italic> since agents share the same policy parameters and update their experiences collectively during training. For instance, Long et al. (<xref ref-type="bibr" rid="B176">Long et al., 2018</xref>) implemented a parallel PPO algorithm to train multiple agents to navigate in simulation. The policy is conditioned on relative goal information and 2D LiDAR data from the past three time steps, which is processed by a 1D CNN. This approach was later validated with real-world scenarios (<xref ref-type="bibr" rid="B83">Fan et al., 2018</xref>). Although agents trained through MARL efficiently learn to avoid collisions with other agents running an identical policy, the approach is often sub-optimal in social navigation contexts, since we assume that all agents exhibit similar behaviors, which may not reflect the diverse and adaptive behaviors in real social interactions.</p>
<p>An alternative to MARL is training navigation policies with simulated crowds. Here, simulated humans exhibit cooperative or reactive behaviors resembling real crowds, enabling agents to adapt to diverse social settings. For instance, <xref ref-type="bibr" rid="B162">Liang et al. (2021)</xref> uses PPO to train agents among cooperative, human-like agents that follow predefined paths and preferred velocities, adjusting their speed based on available space (<xref ref-type="bibr" rid="B213">Narang et al., 2015</xref>). Conversely, <xref ref-type="bibr" rid="B135">Jin et al. (2020)</xref> trains a DDPG-based policy in simulation with non-cooperative, ORCA-modeled humans (<xref ref-type="bibr" rid="B304">Van Den Berg et al., 2011</xref>), who react to obstacles and others without considering the robot&#x2019;s path. The agent&#x2019;s state is captured by multiple 2D LiDAR scans, decoupled from its motion and adjusted for heading differences over time, effectively highlighting dynamic obstacles independently of the robot&#x2019;s motion.</p>
<sec id="s2-1-1">
<label>2.1.1</label>
<title>Learning from demonstration</title>
<p>Imitation learning (IL) enables learning an end-to-end policy directly from expert demonstrations, bypassing the need for hand-crafted rewards. While inverse reinforcement learning (IRL) infers a reward function from human demonstrations or pedestrian datasets (<xref ref-type="bibr" rid="B142">Kim and Pineau, 2016</xref>; <xref ref-type="bibr" rid="B82">Fahad et al., 2018</xref>) then learns a policy, behavioral cloning (BC) learns actions directly from demonstrations but struggles in dynamic settings due to its reliance on fixed data. More advanced IL approaches aim to overcome these limitations. One of the earliest data-driven approaches for static obstacle navigation, proposed by <xref ref-type="bibr" rid="B235">Pfeiffer et al. (2017)</xref>, uses a goal-conditioned model with 1D CNN and pooling layers trained using BC. The model takes in 2D LiDAR readings and goal information to predict actions and is trained on demonstration data collected using the dynamic window approach (DWA) planner (<xref ref-type="bibr" rid="B93">Fox et al., 1997</xref>). While effective in static environments, this approach does not incorporate past observations, reducing its effectiveness in dynamic obstacle scenarios. Similarly, CANet (<xref ref-type="bibr" rid="B175">Long et al., 2017</xref>) applies behavioral cloning to learn a navigation policy from multi-agent data generated using ORCA planner (<xref ref-type="bibr" rid="B304">Van Den Berg et al., 2011</xref>). The model is an MLP trained to output a probability distribution over 61 pre-defined 2D velocity clusters, capturing a range of socially aware navigational behaviors. A value iteration network (VIN)-based planner, proposed by <xref ref-type="bibr" rid="B167">Liu et al. (2018)</xref>, applies VIN (<xref ref-type="bibr" rid="B291">Tamar et al., 2016</xref>) to social navigation. VIN introduces a neural network architecture with a differentiable planning module that approximates the classical value iteration algorithm. Given a reward map and local transition model, VIN iteratively maps rewards and previous value estimates into Q-values using convolutional layers, where each channel corresponds to an action&#x2019;s outcome. A channel-wise max pooling layer retrieves the maximum over actions, yielding the updated value function, which is then used by a greedy reactive policy network (e.g., softmax) to generate an action distribution. <xref ref-type="bibr" rid="B167">Liu et al. (2018)</xref> extend VIN by adding an MLP that combines the VIN output with the robot&#x2019;s velocity to predict actions. Trained in a supervised manner on real and synthetic maps with demonstration actions derived from a reactive optimization-based planner, this approach provides a novel perspective on navigation. However, it&#x2019;s limited to static environments and needs to be extended to dynamic settings with crowds. Another approach, GAIL, is used by <xref ref-type="bibr" rid="B290">Tai et al. (2018)</xref> to train a navigation policy. GAIL employs a generator (policy) that processes depth images to predict actions, while a discriminator distinguishes between the generator&#x2019;s actions and expert demonstrations. To stabilize training, the discriminator is defined as a regression network, inspired by WGAN (<xref ref-type="bibr" rid="B9">Arjovsky et al., 2017</xref>), rather than a standard classifier. Initially, the policy is pre-trained with behavioral cloning on expert data and then fine-tuned using TRPO with the discriminator. The main advantage of GAIL is its use of online simulation-based training, which helps mitigate generalization issues. MuSoHu (<xref ref-type="bibr" rid="B219">Nguyen et al., 2023</xref>) addresses data scarcity in data-driven navigation by providing a large-scale dataset of 100 km of human navigation patterns collected with a helmet-mounted sensor suite. Applying behavioral cloning on this dataset produces a human-like path-planning policy that mitigates behavior modeling inaccuracies and shows strong real-world performance. DeepMoTIon (<xref ref-type="bibr" rid="B116">Hamandi et al., 2019</xref>) aims to mimic human pedestrian behavior by using imitation learning to train a navigation policy. The approach uses pedestrian datasets to simulate human-centric LiDAR data, training an LSTM-based policy through supervised learning. The model predicts the pedestrian&#x2019;s future direction and velocity based on its LiDAR data and final goal. To account for variability in human behavior, it employs a Gaussian distribution for direction prediction, enabling the capture of diverse movement patterns in similar scenarios.</p>
</sec>
<sec id="s2-1-2">
<label>2.1.2</label>
<title>Model-based RL</title>
<p>World models provide agents with internal representations of environment dynamics, enabling more informed, end-to-end decision-making. One prominent example is NavRep (<xref ref-type="bibr" rid="B73">Dugas et al., 2021</xref>), which integrates the World Model framework (<xref ref-type="bibr" rid="B112">Ha and Schmidhuber, 2018</xref>) with the PPO algorithm to train a policy. NavRep introduces <italic>rings</italic>, a novel 2D LiDAR representation that arranges data into exponentially spaced radial intervals within a polar coordinate grid, enhancing close-range resolution. Similarly, <xref ref-type="bibr" rid="B61">Cui et al. (2021)</xref> applies world models with the TD3 algorithm in a MARL framework, with the state represented by stacked 2D obstacle maps generated from multiple LiDAR scans.</p>
</sec>
<sec id="s2-1-3">
<label>2.1.3</label>
<title>Enhanced perception methods</title>
<p>Most methods discussed so far rely on a single sensor input, which can be prone to noise and limited in accuracy. To enhance perception robustness for end-to-end systems, sensor fusion techniques are employed. For example, <xref ref-type="bibr" rid="B162">Liang et al. (2021)</xref> processes 2D LiDAR data using a 1D CNN and depth images using a 2D CNN, with inputs collected over three consecutive time steps, and combines the outputs through concatenation. In another approach, Han et al. (<xref ref-type="bibr" rid="B117">Han Y. et al., 2022</xref>) propose a fusion network that integrates RGB images and 2D LiDAR data to produce depth information. The 2D LiDAR data is first transformed into the camera&#x2019;s coordinate frame, then combined with RGB data through an encoder-decoder CNN network (<xref ref-type="bibr" rid="B182">Ma and Karaman, 2018</xref>) to produce a depth image. The depth image is processed by a self-attention module, which prioritizes pixels based on factors such as robot type, goal position, and velocity, thus enhancing the agent&#x2019;s situational awareness. Some navigation systems focus on optimizing performance with sensors that have limited fields of view. In these setups, self-supervised and supervised approaches are used to improve the agent&#x2019;s situational awareness. For example, <xref ref-type="bibr" rid="B55">Choi et al. (2019)</xref> employ an actor-critic algorithm where the actor network uses an LSTM, while the critic receives additional information, such as a local 2D map. This approach allows the actor to rely on temporal cues, while the critic aids in evaluating action choices more accurately. Similarly, <xref ref-type="bibr" rid="B208">Monaci et al. (2022)</xref> introduce a method where an initial policy is trained using privileged information, such as precise human positions within the environment. This policy is subsequently distilled into a non-privileged policy that learns to approximate the privileged information through supervised learning.</p>
</sec>
<sec id="s2-1-4">
<label>2.1.4</label>
<title>Multi-objective and hierarchical RL</title>
<p>Multi-objective reinforcement learning (MORL) (<xref ref-type="bibr" rid="B257">Roijers et al., 2013</xref>) frameworks are increasingly applied in end-to-end navigation tasks where agents must balance multiple, often conflicting, objectives. MORL allows a policy to be trained on multiple different objectives, enabling the adjustment of objective weightings, referred to as a preference vector, during deployment (<xref ref-type="bibr" rid="B119">Hayes et al., 2022</xref>). This flexibility is particularly beneficial in dynamic social environments, where safety, efficiency, and comfort are key yet sometimes competing. For example, <xref ref-type="bibr" rid="B52">Cheng et al. (2023)</xref> implement a vectorized Q-learning-based MORL algorithm to train a policy with a simulated crowd. Meanwhile, <xref ref-type="bibr" rid="B56">Choi et al. (2020)</xref> use the SAC MORL algorithm to train a navigation policy with a preference vector learned from human feedback, sampled through a Bayesian neural network (<xref ref-type="bibr" rid="B22">Blundell et al., 2015</xref>). Hierarchical reinforcement learning (HRL) divides complex tasks into manageable sub-tasks or sub-goals, allowing an agent to focus on different levels of decision-making. In HRL architectures, the high-level policy selects sub-goals, while the low-level policies execute these sub-goals through specific navigation actions. For instance, <xref ref-type="bibr" rid="B154">Lee et al. (2023)</xref> propose an HRL framework in which the high-level policy focuses on reaching the goal efficiently, minimizing time-to-goal. This policy generates a skill vector, which is then interpreted by the low-level policy to execute specific navigation skills, such as collision avoidance, goal-reaching, and maintaining a safe distance. Both levels of policy utilize 2D LiDAR data and goal state information. Other HRL approaches offer variations in task distribution and shared information. <xref ref-type="bibr" rid="B358">Zhu and Hayashibe (2022)</xref> use a high-level policy as a safety controller to halt the low-level policy if necessary, while <xref ref-type="bibr" rid="B315">Wang et al. (2021)</xref> implement an HRL framework in which the high-level policy shares a sub-goal with the low-level navigation policy.</p>
</sec>
<sec id="s2-1-5">
<label>2.1.5</label>
<title>Vision-based navigation</title>
<p>In vision-based end-to-end navigation, RGB or RGB-D cameras provide input for agents to reach goals specified by relative position (PointGoal), target images (ImageGoal), or instructions (Vision-Language Navigation). These planners excel in visually rich settings without global maps, relying solely on relative goal information. Policies typically use CNN-RNN architectures, where CNNs process images and RNNs build an internal map (<xref ref-type="bibr" rid="B151">Kulh&#xe1;nek et al., 2019</xref>). Even <italic>blind</italic> agents, lacking vision but using memory-based policies, can navigate efficiently via spatial awareness and wall-following strategies (<xref ref-type="bibr" rid="B324">Wijmans et al., 2023</xref>). Such methods use photorealistic simulators based on real-world scans (<xref ref-type="bibr" rid="B37">Chang et al., 2017</xref>) and often employ discrete actions for training efficiency. Vision-based social navigation is emerging, with proximity-aware (<xref ref-type="bibr" rid="B32">Cancelli et al., 2023</xref>) and Falcon (<xref ref-type="bibr" rid="B104">Gong et al., 2024</xref>) methods using auxiliary tasks to better anticipate and navigate around pedestrians and obstacles.</p>
</sec>
<sec id="s2-1-6">
<label>2.1.6</label>
<title>Language models in navigation</title>
<p>Vision-language models (VLMs) are powerful multimodal models with the ability to support navigation through reasoning, visual grounding, and contextual understanding. Early work on vision-language navigation (VLN) (<xref ref-type="bibr" rid="B6">Anderson et al., 2018a</xref>) introduced text-based high-level planning, which can be extended to social navigation for local decision-making (<xref ref-type="bibr" rid="B161">Li et al., 2024</xref>). Beyond high-level planning, several recent hybrid methods integrate VLMs directly into the social navigation pipeline. <xref ref-type="bibr" rid="B281">Song et al. (2024)</xref> use a VLM to select high-level direction and speed, which are integrated with goal and obstacle costs in a model-based planner, with weights determined through an additional VLM prompt. GSON (<xref ref-type="bibr" rid="B180">Luo et al., 2025</xref>) leverages VLMs to detect social groups and integrates the results into an MPC planner to generate paths that avoid them. OLiVia-Nav (<xref ref-type="bibr" rid="B214">Narasimhan et al., 2025</xref>) distills social context from a large VLM into lightweight encoders that provide semantic inputs to a trajectory planner, which then generates candidate motions and selects the one most aligned with captions distilled from expert demonstrations. OLiVia-Nav further incorporates lifelong learning to update its encoders with new data. Related to this, <xref ref-type="bibr" rid="B224">Okunevich et al. (2025)</xref> introduce an online learning approach that adapts a social module in real time, updating the social cost function during deployment. Alternatively, coding-capable large language models (LLMs) have been prompted to generate reward functions from natural language preference descriptions (<xref ref-type="bibr" rid="B183">Ma et al., 2023</xref>), with applications in navigation and preference alignment (<xref ref-type="bibr" rid="B321">Wang et al., 2024</xref>). Social-LLaVA (<xref ref-type="bibr" rid="B232">Payandeh et al., 2024</xref>) leverages a VLM fine-tuned for social robot navigation to directly map decisions onto a predefined set of low-level navigation primitives. Despite this progress, the slow inference and high computational demands of VLMs currently limit their use for real-time reactive social navigation. As a result, they are mostly applied as global planners, semantic encoders, or social-context modules, while their broader potential remains underexplored.</p>
</sec>
<sec id="s2-1-7">
<label>2.1.7</label>
<title>Self-supervised learning</title>
<p>Beyond RL, self-supervised methods enable partial or full training of navigation policies using generated labels. For example, <xref ref-type="bibr" rid="B124">Hoeller et al. (2021)</xref> train a VAE to encode depth data, filter noise, and enhance sim-to-real transfer, providing informative representations for faster RL training. <xref ref-type="bibr" rid="B340">Yang et al. (2023)</xref> propose a bi-level framework where a neural network predicts waypoints optimized through a differentiable ESDF-based cost function; while deployment is simplified by using a spline to fit waypoints. <xref ref-type="bibr" rid="B261">Roth et al. (2024)</xref> further incorporate semantic costmaps, though dynamic obstacle avoidance remains unevaluated.</p>
<p>Overall, end-to-end navigation directly maps sensor inputs to actions and supports continuous actions, multi-agent training, model-based RL, multi-objective and hierarchical frameworks, VLMs, and self-supervised learning. However, challenges remain in ensuring safety and robustness.</p>
</sec>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Human position-based navigation</title>
<p>The challenging nature of collision avoidance in navigation has led to methods that rely on known positions and velocities of dynamic obstacles, such as humans (see <xref ref-type="table" rid="T2">Table 2</xref>). These positions are obtained through a detection and tracking module (see <xref ref-type="sec" rid="s3-3">Section 3.3</xref>), allowing the robot to account for surrounding agents in its navigation decisions. In this setup, the human state is often represented as <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, with the position <inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and velocity <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> defined in the robot frame.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Human position-based social navigation algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Input</th>
<th align="left">Architecture</th>
<th align="left">Algorithm</th>
<th align="left">Output</th>
<th align="left">Training simulator</th>
<th align="left">Real-world demo</th>
<th align="left">Code</th>
<th align="left">Training type</th>
<th align="left">Detection/<break/>Tracking</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CADRL (<xref ref-type="bibr" rid="B42">Chen et al., 2017b</xref>)</td>
<td align="left">Human pos. and vel</td>
<td align="left">FC</td>
<td align="left">Deep V-learning</td>
<td align="left">Sampled <inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CADRL Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">Multi-agent</td>
<td align="left">LiDAR clustering (<xref ref-type="bibr" rid="B31">Campbell et al., 2013</xref>)</td>
</tr>
<tr>
<td align="left">SA-CADRL (<xref ref-type="bibr" rid="B43">Chen et al., 2017c</xref>)</td>
<td align="left">Human pos. and vel</td>
<td align="left">FC</td>
<td align="left">Deep V-learning</td>
<td align="left">Sampled <inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CADRL Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">Multi-agent</td>
<td align="left">LiDAR &#x2b; RGB detection (<xref ref-type="bibr" rid="B201">Miller et al., 2016</xref>) LiDAR tracking (<xref ref-type="bibr" rid="B31">Campbell et al., 2013</xref>)</td>
</tr>
<tr>
<td align="left">GA3C-CADRL (<xref ref-type="bibr" rid="B80">Everett et al., 2018</xref>)</td>
<td align="left">Human pos. and vel</td>
<td align="left">LSTM &#x2b; FC</td>
<td align="left">A3C</td>
<td align="left">Sampled <inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CADRL Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">Multi-agent</td>
<td align="left">LiDAR &#x2b; RGB detection (<xref ref-type="bibr" rid="B201">Miller et al., 2016</xref>) (<xref ref-type="bibr" rid="B166">Liu et al., 2016</xref>), LiDAR tracking (<xref ref-type="bibr" rid="B31">Campbell et al., 2013</xref>)</td>
</tr>
<tr>
<td align="left">DenseCAvoid (<xref ref-type="bibr" rid="B268">Sathyamoorthy et al., 2020a</xref>)</td>
<td align="left">RGB-D &#x2b; 2D LiDAR &#x2b; Human pos. and vel</td>
<td align="left">2D CNN &#x2b; FC &#x2b;1D CNN</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Gazebo Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Random non- cooperative crowd</td>
<td align="left">YOLOv3 detection (<xref ref-type="bibr" rid="B250">Redmon, 2018</xref>) RobustTP prediction (<xref ref-type="bibr" rid="B36">Chandra et al., 2019</xref>)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B118">Han et al. (2022b)</xref>
</td>
<td align="left">RVO 6D vector &#x2b; Distance&#x2b; Reciprocal collision time</td>
<td align="left">BiGRU &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">Demo (Robot-Robot) &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">Multi-agent</td>
<td align="left">None</td>
</tr>
<tr>
<td align="left">DRL-VO (<xref ref-type="bibr" rid="B328">Xie and Dames, 2023</xref>)</td>
<td align="left">2D LiDAR &#x2b; Human pos. and vel</td>
<td align="left">2D CNN &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Gazebo Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">SFM crowd (<xref ref-type="bibr" rid="B103">Gloor, 2016</xref>)</td>
<td align="left">YOLOv3 detection (<xref ref-type="bibr" rid="B250">Redmon, 2018</xref>) &#x2b; MHT tracking (<xref ref-type="bibr" rid="B345">Yoon et al., 2018</xref>)</td>
</tr>
<tr>
<td align="left">ILPP (<xref ref-type="bibr" rid="B245">Qin et al., 2021</xref>)</td>
<td align="left">Human position and velocity &#x2b;2D LiDAR &#x2b; Global path</td>
<td align="left">2D CNN &#x2b; FC &#x2b; Attn Block</td>
<td align="left">BC</td>
<td align="left">Confidence Map (guides A&#x2a; planner)</td>
<td align="left">None</td>
<td align="left">Demo</td>
<td align="left">No</td>
<td align="left">Data-driven</td>
<td align="left">LiDAR Detection &#x2b; Kalman filter and Hungarian Algorithm</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B67">De Heuvel et al. (2022)</xref>
</td>
<td align="left">Human pos. and heading</td>
<td align="left">FC</td>
<td align="left">TD3</td>
<td align="left">
<inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Pybullet Simulator</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Static human</td>
<td align="left">None</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B68">De Heuvel et al. (2023)</xref>
</td>
<td align="left">Depth image &#x2b; Human pos</td>
<td align="left">VAE (CNN) &#x2b; LSTM &#x2b; FC</td>
<td align="left">TD3</td>
<td align="left">
<inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">iGibson simulator</td>
<td align="left">None</td>
<td align="left">No</td>
<td align="left">Static human</td>
<td align="left">None</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B189">Marta et al. (2023)</xref>
</td>
<td align="left">2D LiDAR (labeled rays) &#x2b; Human pos. and vel</td>
<td align="left">FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<break/>&#x2b; SFM force</td>
<td align="left">Custom Unity Sim</td>
<td align="left">None</td>
<td align="left">No</td>
<td align="left">SFM crowd</td>
<td align="left">None</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B69">De Heuvel et al. (2024)</xref>
</td>
<td align="left">2D LiDAR &#x2b; Human pos</td>
<td align="left">FC</td>
<td align="left">MORL-TD3</td>
<td align="left">
<inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">iGibson sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Static human</td>
<td align="left">None</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>A foundational approach in human position-based navigation is Collision Avoidance with Deep Reinforcement Learning (CADRL), introduced by <xref ref-type="bibr" rid="B42">Chen et al. (2017b)</xref>. CADRL uses a model-based RL framework to learn a value function over the joint state space of the robot and surrounding agents. The optimal action is derived as <inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="script">A</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="script">T</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="script">A</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is a set of sampled actions, <inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the learned value function, and <inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="script">T</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> represents the estimated transition dynamics. In CADRL, the transition dynamics model for humans is estimated using a simplified constant velocity model.</p>
<p>Building on CADRL, Socially Aware CADRL (SA-CADRL) (<xref ref-type="bibr" rid="B43">Chen et al., 2017c</xref>) incorporates social norms, such as overtaking, directly into the reward function. The value function in SA-CADRL is computed over a fixed set of agents and is trained similarly to CADRL using the multi-agent reinforcement learning (MARL) framework. Further advancements, such as GA3C-CADRL (<xref ref-type="bibr" rid="B80">Everett et al., 2018</xref>), extend SA-CADRL by applying the A3C algorithm and integrating an LSTM layer, enabling the policy to process an arbitrary number of agents as input, thereby increasing scalability in crowded environments. Additionally, GA3C-CADRL simplifies the reward structure by removing explicit social norms. Further research by <xref ref-type="bibr" rid="B81">Everett et al. (2021)</xref> explores the impact of the LSTM on this model&#x2019;s performance in complex, multi-agent scenarios. While GA3C-CADRL performs well, using an LSTM to encode multiple agents may affect consistency due to LSTM&#x2019;s sensitivity to input order.</p>
<p>A range of methods leverage the concept of velocity obstacles (VO) in state or reward functions to promote collision avoidance in navigation policies. <xref ref-type="bibr" rid="B118">Han R. et al. (2022)</xref> propose an RL policy that uses reciprocal velocity obstacles (RVO) (<xref ref-type="bibr" rid="B303">Van den Berg et al., 2008</xref>) to model agent interactions. The policy processes RVO parameters, including a 6D vector (preferred velocity and boundary velocities), distance, and reciprocal collision time for each human, using a bi-directional RNN (BiGRU). The reward function penalizes overlapping RVO areas. Some approaches, such as DRL-VO (<xref ref-type="bibr" rid="B328">Xie and Dames, 2023</xref>) and DenseCAvoid (<xref ref-type="bibr" rid="B268">Sathyamoorthy et al., 2020a</xref>), incorporate both human positions and sensor data to handle static obstacle avoidance in navigation. DRL-VO combines human positions with 2D LiDAR data, leveraging a VO-based reward function to encourage collision-free trajectories. This fusion of human position data with LiDAR enables effective static and dynamic obstacle avoidance. Similarly, DenseCAvoid uses the PPO algorithm to train a policy that fuses 2D LiDAR and RGB-D data for enhanced static obstacle detection. Building on an architecture similar to <xref ref-type="bibr" rid="B162">Liang et al. (2021)</xref>, DenseCAvoid integrates single-step human motion predictions using RobustTP (<xref ref-type="bibr" rid="B36">Chandra et al., 2019</xref>), enabling the model to anticipate human movements in dynamic environments.</p>
<p>ILPP (<xref ref-type="bibr" rid="B245">Qin et al., 2021</xref>) applies imitation learning to generate a navigation confidence map that modifies the global path to incorporate collision avoidance. To produce a confidence map, the model takes LiDAR data, global path, pedestrian positions and velocities, and robot odometry. Additionally, ILPP predicts when global re-planning is necessary, especially if the expert path deviates from the global path. The model is trained using 1.3 h of a human driver operating a motorized wheelchair. To derive a path from the confidence map, the destination is set where the goal path meets the grid edge, and an A&#x2a; planner finds the lowest-cost route to the destination, which is then smoothed using Gaussian filtering before being executed by a low-level controller.</p>
<sec id="s2-2-1">
<label>2.2.1</label>
<title>Preference-aware navigation</title>
<p>Approaches that incorporate human demonstrations and preferences into policy training have proven effective for aligning robot behavior with human expectations in social navigation. <xref ref-type="bibr" rid="B67">De Heuvel et al. (2022)</xref> use the SAC algorithm with behavioral cloning to train a policy in simulation, closely fitting human demonstration trajectories collected via a VR pointer. This work is extended in <xref ref-type="bibr" rid="B68">De Heuvel et al. (2023)</xref> by adding a perception pipeline that predicts future human positions. Building on this, <xref ref-type="bibr" rid="B69">De Heuvel et al. (2024)</xref> employ MORL-TD3 with multiple objectives, including a human demonstration distilled into a reward function using D-REX. Lastly, <xref ref-type="bibr" rid="B189">Marta et al. (2023)</xref> adopt a multi-objective approach to balance an expert-designed objective with a human preference objective derived from a reward model trained on pairwise human trajectory comparisons.</p>
<p>Overall, human position-based navigation utilizes explicit knowledge of human positions and velocities to enable safer and more socially-aware navigation policies. Techniques such as CADRL-based methods establish foundational frameworks by learning interaction-aware value functions. Moreover, incorporating human preferences and demonstrations ensures policies align closely with human expectations.</p>
</sec>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>Human attention-based navigation</title>
<p>Human attention-based navigation approaches explicitly model the attention between humans within a crowd. Human Attention-based approaches have become a key component in social navigation, enabling policies that adapt to both individual and crowd dynamics, and achieving significant performance improvement (see <xref ref-type="table" rid="T3">Table 3</xref>). These methods explicitly model relationships between human features using pooling layers or graph neural networks (GNNs) to represent mutual influences. Pooling layers provide a compact, unified representation of human features, which, when combined with individual features, encodes human-human attention. In graph-based approaches, the robot and humans are nodes in the input graph, generating node embeddings that capture human-human and robot-human relationships.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Human-human interaction-based social navigation algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Input</th>
<th align="left">Architecture</th>
<th align="left">Algorithm</th>
<th align="left">Output</th>
<th align="left">Training sim</th>
<th align="left">Real-world demo</th>
<th align="left">Code</th>
<th align="left">Detection/<break/>Tracking</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SARL (<xref ref-type="bibr" rid="B47">Chen et al., 2019b</xref>)</td>
<td align="left">Human pos. and vel</td>
<td align="left">Self-Attn &#x2b; LSTM &#x2b; FC</td>
<td align="left">Deep V-learning</td>
<td align="left">80 discrete <inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">Depth projection detection</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B168">Liu et al. (2020a)</xref>
</td>
<td align="left">2D LiDAR grid/map &#x2b; Human pos. and vel</td>
<td align="left">Self-Attn &#x2b; LSTM &#x2b;2D CNN &#x2b; FC</td>
<td align="left">A3C</td>
<td align="left">Discrete <inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">MobileNetSSD &#x2b; Block matching detection (<xref ref-type="bibr" rid="B78">Eppenberger et al., 2020</xref>)</td>
</tr>
<tr>
<td align="left">NaviGAN (<xref ref-type="bibr" rid="B299">Tsai and Oh, 2020</xref>)</td>
<td align="left">Human path</td>
<td align="left">GAN &#x2b; LSTM &#x2b; Pooling &#x2b; FC</td>
<td align="left">GAN</td>
<td align="left">
<inline-formula id="inf37">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">None</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">LiDAR &#x2b; Kalman filter</td>
</tr>
<tr>
<td align="left">DS-RNN (<xref ref-type="bibr" rid="B170">Liu et al., 2021</xref>)</td>
<td align="left">Human position</td>
<td align="left">S-RNN (LSTM/GRU) &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf38">
<mml:math id="m38">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">YOLOv3 detection (<xref ref-type="bibr" rid="B250">Redmon, 2018</xref>) &#x2b; DeepSORT tracking (<xref ref-type="bibr" rid="B325">Wojke et al., 2017</xref>)</td>
</tr>
<tr>
<td align="left">GazeNav (<xref ref-type="bibr" rid="B48">Chen et al., 2020a</xref>)</td>
<td align="left">Human pos. and vel</td>
<td align="left">GCN &#x2b; FC</td>
<td align="left">Deep V-learning</td>
<td align="left">Sampled <inline-formula id="inf39">
<mml:math id="m39">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">None</td>
<td align="left">No</td>
<td align="left">None</td>
</tr>
<tr>
<td align="left">Navistar (<xref ref-type="bibr" rid="B319">Wang et al., 2023a</xref>)</td>
<td align="left">Human pos. and vel</td>
<td align="left">Multi-Head Attn &#x2b; GCN &#x2b; FC</td>
<td align="left">SAC</td>
<td align="left">
<inline-formula id="inf40">
<mml:math id="m40">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">Lab experiment &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">Velocity predictor (YOLO &#x2b; DeepSORT) (<xref ref-type="bibr" rid="B240">Pramanik et al., 2021</xref>) &#x2b; Distance estimator (<xref ref-type="bibr" rid="B17">Bertoni et al., 2021</xref>)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B171">Liu et al. (2023a)</xref>
</td>
<td align="left">Human occupancy grid</td>
<td align="left">GAT &#x2b; 2D CNN &#x2b; LSTM &#x2b; FC</td>
<td align="left">DQN</td>
<td align="left">5 discrete <inline-formula id="inf41">
<mml:math id="m41">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Custom Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
<td align="left">Yolov3 detection (<xref ref-type="bibr" rid="B250">Redmon, 2018</xref>) &#x2b; pointcloud tracking (<xref ref-type="bibr" rid="B169">Liu et al., 2020b</xref>)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>SARL (<xref ref-type="bibr" rid="B47">Chen et al., 2019b</xref>) builds on CADRL (<xref ref-type="bibr" rid="B42">Chen et al., 2017b</xref>) by introducing an attention and a pooling module to explicitly capture human-human attention. The attention module encodes features of each human relative to surrounding humans using a human-centered local map. In this local map, each human&#x2019;s surrounding individuals are divided into grid cells concatenated with the human and robot states, then the features are passed into an MLP to produce a human embedding vector. To capture human-human attention and transform an arbitrary number of human embeddings into a fixed-size vector, SARL uses a self-attention pooling module, an attention mechanism adapted from Transformers. This attention mechanism assigns scalar weights to each human embedding vector and computes a unified output by summing the weighted embeddings across all humans. This dual-stage position-based encoding via the local map and self-attention pooling improves social navigation performance compared to methods without explicit attention encoding, though local maps offered a slight performance improvement during testing. During deployment, SARL may also be adapted to use a single-step human trajectory prediction model to estimate the next state, offering a more accurate alternative to the constant velocity model used in CADRL.</p>
<p>SOADRL (<xref ref-type="bibr" rid="B168">Liu et al., 2020a</xref>) extends SARL to a model-free RL setup, introducing a two-policy switching mechanism to address both dynamic and static obstacles. When humans are present, SOADRL combines SARL&#x2019;s output with a robot-centric angular map or 2D occupancy grid for static obstacle encoding. In the absence of humans, SOADRL switches to a policy that relies solely on the map input, ensuring efficient navigation through static obstacles.</p>
<p>NaviGAN (<xref ref-type="bibr" rid="B299">Tsai and Oh, 2020</xref>) introduces a learning-based social force model (SFM) for navigation using a dual LSTM-based GAN architecture. The model&#x2019;s first LSTM generates an intention force based on the robot&#x2019;s goal and past state sequence, while the second LSTM generates a social force that accounts for human interactions. It uses a pooling layer similar to the one in Social-GAN (<xref ref-type="bibr" rid="B110">Gupta et al., 2018</xref>) to encode human history. It also incorporates a fluctuation force for randomness. The combined intention and social forces determine the robot&#x2019;s future actions. A discriminator is used during training to encourage realistic behavior, distinguishing between generated actions and expert actions from a real-world pedestrian dataset. To incorporate temporal information, DS-RNN (<xref ref-type="bibr" rid="B170">Liu et al., 2021</xref>) uses a three-RNN architecture trained with PPO for social navigation. One RNN encodes each human&#x2019;s past positions relative to the robot; another encodes the robot&#x2019;s past velocities. These embeddings are combined via attention pooling (without modeling human-human attentions) and, along with the robot&#x2019;s state, fed into a third RNN that outputs the policy action and value function.</p>
<sec id="s2-3-1">
<label>2.3.1</label>
<title>Graph neural network-based navigation</title>
<p>GazeNav (<xref ref-type="bibr" rid="B48">Chen et al., 2020a</xref>) employs a model-based RL approach with gaze-based attention that uses 2 two-layer Graph Convolutional Networks (GCNs) to define its value function. The first GCN, an attention network, treats the robot and humans as graph nodes with uniform edge weights, predicting attention weights for each connection. The second GCN is an aggregation network that uses the predicted attention weights as edge values to compute embedding vectors for each human-robot pair, which are then passed into an MLP-based value function. To train the attention network, GazeNav introduces three supervised methods: uniform weights, distance-based weights, and gaze-modulated weights. The gaze-modulated weights are obtained by tracking human gaze in a simulated environment, assigning higher attention to humans within the gaze direction. Experiments show that gaze-modulated weights outperform uniform, distance-based, and self-attention-based weights (<xref ref-type="bibr" rid="B47">Chen C. et al., 2019</xref>), demonstrating the benefits of incorporating human gaze data. For a more expressive representation, Navistar (<xref ref-type="bibr" rid="B319">Wang W. et al., 2023</xref>) uses a three-block architecture to model spatio-temporal crowd interactions. A spatial block (GCN plus multi-head attention) creates spatial embeddings; a temporal block applies multi-head attention with positional encoding for each human. A multi-modal transformer block then merges these outputs using cross-attention and self-attention to produce the final action and value outputs. In a related approach, <xref ref-type="bibr" rid="B171">Liu Z. et al. (2023)</xref> integrate GNNs with occupancy grids to capture spatial-temporal characteristics. At each time step, the environment is divided into a robot-centered grid and an obstacle-centered grid for each human, both processed by a CNN. The CNN outputs are then passed through an LSTM to capture temporal patterns, feeding into a Graph Attention Network (GAT) that produces interaction-aware embeddings. The control policy uses an MLP to generate action distributions from the GAT&#x2019;s aggregated output.</p>
<p>To summarize, human attention-based navigation methods explicitly model human-human and human-robot attentions to enable socially-aware and adaptive policies. Approaches utilizing pooling layers, GNNs and RNNs, provide improved social compliance by capturing spatial and temporal relationships.</p>
</sec>
</sec>
<sec id="s2-4">
<label>2.4</label>
<title>Human prediction-based navigation</title>
<p>Human Prediction-based Social Navigation (see <xref ref-type="table" rid="T4">Table 4</xref>) leverages human trajectory prediction to enable more strategic, optimal navigation in dynamic environments (see <xref ref-type="sec" rid="s3-3-2">Section 3.3.2</xref>). This approach aligns with model-based RL principles, where the human prediction model serves as a dynamics model, guiding decision-making by simulating future states. To leverage this predictive capability, the navigation system should plan over a similar multi-second horizon rather than just single-step actions. Early work in this area applied techniques like Monte Carlo Tree Search (MCTS) for high-level decision-making in autonomous vehicles (<xref ref-type="bibr" rid="B231">Paxton et al., 2017</xref>) and optimization-based planners such as MPC for robots (<xref ref-type="bibr" rid="B91">Finn and Levine, 2017</xref>). One notable example is <xref ref-type="bibr" rid="B45">Chen et al. (2018)</xref>, who use a Social-LSTM (<xref ref-type="bibr" rid="B2">Alahi et al., 2016</xref>) to predict human trajectories, incorporating this into an optimization-based timed elastic band (TEB) planner (<xref ref-type="bibr" rid="B258">R&#xf6;smann et al., 2015</xref>) with adaptive travel modes that adjust based on crowd density and movement direction.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Human prediction-based social navigation algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Input</th>
<th align="left">Architecture</th>
<th align="left">Algorithm</th>
<th align="left">Output</th>
<th align="left">Training sim</th>
<th align="left">Real-world demo</th>
<th align="left">Code</th>
<th align="left">Detection/<break/>Tracking</th>
<th align="left">Prediction</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">MCTS-RNN (<xref ref-type="bibr" rid="B76">Eiffertet al., 2020a</xref>)</td>
<td align="left">Human path</td>
<td align="left">LSTM Enc-Dec &#x2b; FC</td>
<td align="left">MCTS (Model-based)</td>
<td align="left">Path <inline-formula id="inf42">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">None</td>
<td align="left">Learned RNN</td>
</tr>
<tr>
<td align="left">MP-RGL (<xref ref-type="bibr" rid="B49">Chen et al., 2020b</xref>)</td>
<td align="left">Human pos. and vel.</td>
<td align="left">GCN &#x2b; FC</td>
<td align="left">MCTS (Model-based)</td>
<td align="left">Discrete paths<break/>
<inline-formula id="inf43">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">YOLO detection &#x2b; EKF tracking</td>
<td align="left">Learned GCN</td>
</tr>
<tr>
<td align="left">GO-MPC (<xref ref-type="bibr" rid="B25">Brito et al., 2021</xref>)</td>
<td align="left">Human pos. and vel.</td>
<td align="left">LSTM &#x2b; FC</td>
<td align="left">PPO &#x2b; MPC</td>
<td align="left">
<inline-formula id="inf44">
<mml:math id="m44">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CADRL Sim</td>
<td align="left">None</td>
<td align="left">Yes</td>
<td align="left">None</td>
<td align="left">Constant velocity model</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B239">Poddar et al. (2023)</xref>
</td>
<td align="left">Human path</td>
<td align="left">LSTM &#x2b; GAN &#x2b; FC</td>
<td align="left">MPC (Model-based)</td>
<td align="left">Discrete <inline-formula id="inf45">
<mml:math id="m45">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">Lab experiment &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">None</td>
<td align="left">S-GAN (<xref ref-type="bibr" rid="B110">Gupta et al., 2018</xref>)</td>
</tr>
<tr>
<td align="left">SARL-SGAN-KCE (<xref ref-type="bibr" rid="B158">Li et al., 2020</xref>)</td>
<td align="left">Human pos. and vel.</td>
<td align="left">GAN &#x2b; LSTM &#x2b; Self-Attn</td>
<td align="left">Deep V-Learning</td>
<td align="left">
<inline-formula id="inf46">
<mml:math id="m46">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">None</td>
<td align="left">No</td>
<td align="left">None</td>
<td align="left">S-GAN (<xref ref-type="bibr" rid="B110">Gupta et al., 2018</xref>)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B172">Liu et al. (2023b)</xref>
</td>
<td align="left">Human path</td>
<td align="left">Multi-Head-Attn &#x2b; GRU &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf47">
<mml:math id="m47">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">CrowdNav</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
<td align="left">DR-SPAAM (<xref ref-type="bibr" rid="B134">Jia et al., 2020</xref>)</td>
<td align="left">GST (<xref ref-type="bibr" rid="B128">Huang et al., 2021</xref>)</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s2-4-1">
<label>2.4.1</label>
<title>MCTS-based navigation</title>
<p>MCTS-RNN (<xref ref-type="bibr" rid="B76">Eiffert et al., 2020a</xref>) is a model-based RL navigation system that uses an LSTM encoder-decoder human prediction model as its dynamics model. The LSTM model is trained on pedestrian datasets and outputs a Gaussian distribution over future human states. Planning is conducted using MCTS with a receding horizon, performing single-step rollouts from each node to reduce runtime, which increases state uncertainty. To handle this, the reward function includes both goal proximity and prediction uncertainty. MP-RGL (<xref ref-type="bibr" rid="B49">Chen C. et al., 2020</xref>) integrates MCTS planning with a GCN-based human prediction model. The GCN operates on a fully connected graph comprising humans and the robot, where edge weights are computed using Gaussian similarity in the node embedding space (<xref ref-type="bibr" rid="B313">Wang X. et al., 2018</xref>). Planning is performed through a simplified MCTS (<xref ref-type="bibr" rid="B222">Oh et al., 2017</xref>), with a <inline-formula id="inf48">
<mml:math id="m48">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-step planning horizon, leveraging value function estimates instead of explicit rollouts.</p>
</sec>
<sec id="s2-4-2">
<label>2.4.2</label>
<title>MPC-based navigation</title>
<p>GO-MPC (<xref ref-type="bibr" rid="B25">Brito et al., 2021</xref>) is a hybrid framework that integrates RL and nonlinear MPC for navigation, where an LSTM-based RL model proposes sub-goals (as Gaussians) and the MPC computes optimal, collision-free trajectories to these sub-goals. The RL model is first supervised-trained with MPC-generated labels, then fine-tuned with PPO, aiming to maximize goal-reaching and minimize collisions. The MPC minimizes distance and control costs, enforcing constraints to avoid predicted human paths. <xref ref-type="bibr" rid="B239">Poddar et al. (2023)</xref> propose a hybrid approach that integrates a Social-GAN (<xref ref-type="bibr" rid="B110">Gupta et al., 2018</xref>) human prediction model with an MPC planner. This approach uses discrete MPC to optimize a cost function that balances goal distance, social distance, and alignment with Social-GAN predictions to encourage human-like behavior. While Social-GAN can generate multiple predictions per human, results indicate that single and multiple prediction scenarios perform comparably to simpler constant-velocity estimates.</p> <p>SARL-SGAN-KCE (<xref ref-type="bibr" rid="B158">Li et al., 2020</xref>) combines Social-GAN predictions with the SARL model (<xref ref-type="bibr" rid="B47">Chen C. et al., 2019</xref>) to choose optimal single-step actions. To ensure smooth motion, the planner constrains the action space by limiting angular velocity and penalizing rapid acceleration changes. Experimental results show that a higher number of trajectory predictions per human achieves performance comparable to a lower number of predictions. Finally, <xref ref-type="bibr" rid="B172">Liu S. et al. (2023)</xref> propose a model-free PPO RL approach that incorporates off-the-shelf human prediction models like GST (<xref ref-type="bibr" rid="B128">Huang et al., 2021</xref>). Human predictions are processed with multi-head human-human attention, then through robot-human attention with the robot&#x2019;s state, followed by a GRU that outputs the value and action. The reward penalizes intersecting predicted human paths, reducing collision risk despite prediction uncertainty.</p>
<p>In summary, human prediction-based navigation enhances decision-making by anticipating future human movements, enabling more strategic and socially compliant planning. Challenges include managing uncertainty from the robot&#x2019;s impact on human behavior and the computational cost of tree-based methods like MCTS, which require repeated action sampling and forward simulation.</p>
</sec>
</sec>
<sec id="s2-5">
<label>2.5</label>
<title>Safety-aware navigation</title>
<p>Considering that learning-based approaches are, in some sense, viewed as black-box methods, researchers have attempted to embed safety and functionality through purposefully designed algorithms (see <xref ref-type="table" rid="T5">Table 5</xref>). These approaches are classified as safety-aware when they introduce an additional module, training strategy, or feature primarily dedicated to safety.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Safety-aware social navigation algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Input</th>
<th align="left">Architecture</th>
<th align="left">Algorithm</th>
<th align="left">Output</th>
<th align="left">Safety mechanism</th>
<th align="left">Training sim</th>
<th align="left">Real-world demo</th>
<th align="left">Code</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B286">Sun et al. (2019)</xref>
</td>
<td align="left">2D LiDAR (labeled rays)</td>
<td align="left">LSTM &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf49">
<mml:math id="m49">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Multi-policy (VO-based switch)</td>
<td align="left">Unity Custom Sim</td>
<td align="left">None</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B140">Katyal et al. (2020)</xref>
</td>
<td align="left">Human path</td>
<td align="left">VAE &#x2b; LSTM &#x2b; FC</td>
<td align="left">Deep V-Learning</td>
<td align="left">2D velocity</td>
<td align="left">Multi-policy (uncertainty switch)</td>
<td align="left">CrowdNav</td>
<td align="left">None</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B84">Fan et al. (2020)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">1D CNN &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf50">
<mml:math id="m50">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Multi-policy (distance switch)</td>
<td align="left">Stage Sim</td>
<td align="left">Lab experiment &#x2b; Video</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B164">Linh et al. (2022)</xref>
</td>
<td align="left">Static and dynamic 2D LiDAR</td>
<td align="left">2D CNN &#x2b; FC</td>
<td align="left">A3C</td>
<td align="left">
<inline-formula id="inf51">
<mml:math id="m51">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Multi-policy (RL policy switch)</td>
<td align="left">Arena-Rosnav</td>
<td align="left">None</td>
<td align="left">Yes</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B220">Nishimura and Yonetani (2020)</xref>
</td>
<td align="left">Human pos. and vel.</td>
<td align="left">Self-Attn &#x2b; FC</td>
<td align="left">Deep V-Learning</td>
<td align="left">8 discrete <inline-formula id="inf52">
<mml:math id="m52">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<break/>&#x2b; Clearing beep</td>
<td align="left">Crowd clearing beep</td>
<td align="left">Custom Sim</td>
<td align="left">None</td>
<td align="left">Yes</td>
</tr>
<tr>
<td align="left">IAN (<xref ref-type="bibr" rid="B72">Dugas et al., 2020</xref>)</td>
<td align="left">Grid map</td>
<td align="left">None</td>
<td align="left">MCTS</td>
<td align="left">
<inline-formula id="inf53">
<mml:math id="m53">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Multi-policy (MCTS switch)</td>
<td align="left">Custom Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B181">L&#xfc;tjens et al. (2019)</xref>
</td>
<td align="left">Human pos. and vel.</td>
<td align="left">LSTM &#x2b; FC</td>
<td align="left">MPC</td>
<td align="left">11 discrete <inline-formula id="inf54">
<mml:math id="m54">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Uncertainty collision prediction</td>
<td align="left">Custom Sim</td>
<td align="left">None</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B269">Sathyamoorthy et al. (2020b)</xref>
</td>
<td align="left">2D LiDAR &#x2b; human pos</td>
<td align="left">1D CNN &#x2b; FC</td>
<td align="left">PPO</td>
<td align="left">
<inline-formula id="inf55">
<mml:math id="m55">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Heading Safety filter</td>
<td align="left">Gazebo Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">XAI-N (<xref ref-type="bibr" rid="B260">Roth et al., 2021</xref>)</td>
<td align="left">2D LiDAR</td>
<td align="left">Decision Tree</td>
<td align="left">PPO &#x2b; VIPER</td>
<td align="left">Discrete <inline-formula id="inf56">
<mml:math id="m56">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Decision Tree</td>
<td align="left">Gazebo Sim</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B132">Jang and Ghaffari (2024)</xref>
</td>
<td align="left">Human pos. and vel.</td>
<td align="left">None</td>
<td align="left">MPC</td>
<td align="left">
<inline-formula id="inf57">
<mml:math id="m57">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Control-Barrier Function</td>
<td align="left">2D sim</td>
<td align="left">None</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">CASRL (<xref ref-type="bibr" rid="B357">Zhou et al., 2023</xref>)</td>
<td align="left">Human pos. and vel.</td>
<td align="left">HGAT &#x2b; MLP</td>
<td align="left">TD3</td>
<td align="left">
<inline-formula id="inf58">
<mml:math id="m58">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Multi-task</td>
<td align="left">CrowdNav</td>
<td align="left">Demo</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B236">Pfeiffer et al. (2018)</xref>
</td>
<td align="left">2D LiDAR</td>
<td align="left">FC</td>
<td align="left">CPO</td>
<td align="left">
<inline-formula id="inf59">
<mml:math id="m59">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Collision avoidance constraint</td>
<td align="left">Custom Sim</td>
<td align="left">Demo</td>
<td align="left">Yes</td>
</tr>
<tr>
<td align="left">SoNIC (<xref ref-type="bibr" rid="B342">Yao et al., 2024</xref>)</td>
<td align="left">Human pos. and vel.</td>
<td align="left">GRU &#x2b; MLP</td>
<td align="left">PPO-Lag</td>
<td align="left">
<inline-formula id="inf60">
<mml:math id="m60">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Constraint</td>
<td align="left">CrowdNav</td>
<td align="left">Demo &#x2b; Video</td>
<td align="left">Yes</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s2-5-1">
<label>2.5.1</label>
<title>Multi-policy navigation</title>
<p>Hybrid multi-policy planning combines multiple strategies, where robots switch policies based on context and uncertainty. For example, <xref ref-type="bibr" rid="B286">Sun et al. (2019)</xref> switches between RL and RVO when a collision is imminent. <xref ref-type="bibr" rid="B140">Katyal et al. (2020)</xref> build on this with risk-averse and aggressive policies. By default, the system follows the aggressive policy but switches to the risk-averse policy in novel social scenarios, identified by an LSTM-based probabilistic pedestrian prediction module that uses goal intent prediction to generate a set of possible trajectories. The policy selector computes uncertainty from these predictions, with higher uncertainty indicating unfamiliar situations where the risk-averse policy is preferred. Extending this approach, <xref ref-type="bibr" rid="B84">Fan et al. (2020)</xref> develop a three-policy system with a scenario classifier to switch between a PID controller, a standard RL policy (<xref ref-type="bibr" rid="B176">Long et al., 2018</xref>), and a safe RL policy with clipped velocity. The classifier relies on two parameters, the <italic>safe radius</italic> and <italic>risk radius</italic>, based on the distance to nearby obstacles. When within the safe radius, the PID policy is used. In the risk radius, the RL policy takes over, and outside both, the safe policy is employed. To address more complex scenarios, Amano and Kato (<xref ref-type="bibr" rid="B4">Amano and Kato, 2022</xref>) add a fourth policy to this setup, a reset policy to move the robot toward a larger unoccupied space if it detects a freezing robot scenario. This extension ensures the robot can navigate out of potentially freezing situations. Furthermore, <xref ref-type="bibr" rid="B164">Linh et al. (2022)</xref> propose a multi-policy system with three policies, using an RL-based policy selector to choose the most appropriate policy dynamically. Policies include both learning-based (RL) and model-based (TEB) planners (<xref ref-type="bibr" rid="B258">R&#xf6;smann et al., 2015</xref>). The selector is trained to optimize rewards by picking the best policy for a given context, combining flexibility with performance for complex navigation tasks.</p>
<p>
<xref ref-type="bibr" rid="B220">Nishimura and Yonetani (2020)</xref> introduce Learning-to-Balance (L2B), a single-policy RL system that dynamically switches between two behaviors: passive crowd avoidance or active path-clearing through audible signals. The robot action is defined by a velocity vector and a binary mode indicator, with a reward function that discourages excessive path-clearing while promoting social distancing. To simulate the impact of path-clearing sounds on human behavior during training, L2B uses a simplified version of emotional reciprocal velocity obstacles (ERVO) (<xref ref-type="bibr" rid="B331">Xu M. et al., 2019</xref>), which accounts for emotional reactions to perceived threats. IAN (<xref ref-type="bibr" rid="B72">Dugas et al., 2020</xref>) is a multi-policy navigation system that uses Monte Carlo Tree Search (MCTS) to choose among three planning policies: intend (RVO planner (<xref ref-type="bibr" rid="B3">Alonso-Mora et al., 2013</xref>) for reactive avoidance), say (verbal path announcement with lower speed and assumed human cooperation), and nudge (DWA planner (<xref ref-type="bibr" rid="B93">Fox et al., 1997</xref>) for cautious progress). MCTS evaluates paths by crowdedness, perceptivity, and permissivity, selecting the lowest-cost route and adapting plans based on each policy&#x2019;s success probability. Both L2B and IAN require the robot to have a speaker and operate where its audio signals are audible.</p>
<p>
<xref ref-type="bibr" rid="B181">L&#xfc;tjens et al. (2019)</xref> propose a hybrid safe RL system based on discrete MPC, optimizing a cost function that accounts for estimated goal-reaching time and predicted collision probability. An ensemble of LSTMs predicts collision probabilities of motion primitives, with MC-dropout (<xref ref-type="bibr" rid="B96">Gal and Ghahramani, 2016</xref>) used for uncertainty estimation. The collision prediction model is trained as a binary classifier in simulation, penalizing uncertainty to encourage safe exploration. However, this approach heavily depends on collision model accuracy, and inaccuracies can lead to overly conservative behavior. <xref ref-type="bibr" rid="B269">Sathyamoorthy et al. (2020b)</xref> introduce Frozone, which prevents robot freezing by detecting potential freezing zones (PFZs) using pedestrian positions and velocities. A convex hull is constructed around predicted pedestrian locations, and the robot computes a deviation angle to avoid these regions. However, in confined spaces like corridors, Frozone may lead the robot toward other obstacles. XAI-N (<xref ref-type="bibr" rid="B260">Roth et al., 2021</xref>) leverages decision trees to create an interpretable navigation policy. XAI-N distills an RL policy (<xref ref-type="bibr" rid="B83">Fan et al., 2018</xref>) into a single decision tree using the VIPER method (<xref ref-type="bibr" rid="B14">Bastani et al., 2018</xref>), prioritizing modifiability and transparency over continuous action control. To enhance performance, the approach incorporates decision rules to address safety challenges such as freezing and oscillation, making it a more reliable option for social navigation.</p>
<p>
<xref ref-type="bibr" rid="B13">Bansal et al. (2020)</xref> propose a Hamilton&#x2013;Jacobi reachability-based framework that augments the human state with a belief over future intent, producing a forward reachable set that includes all likely pedestrian states for fixed time-horizon with probability above threshold <inline-formula id="inf61">
<mml:math id="m61">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. These sets are incorporated as time-dependent obstacles and avoided with a spline-based trajectory planner. However, in highly populated scenes, the predicted reachable sets may overlap heavily for a low threshold, effectively blocking all routes and causing the robot to freeze, and while the human model parameters can be learned from data, the overall prediction approach is model-based. <xref ref-type="bibr" rid="B132">Jang and Ghaffari (2024)</xref> learn social zones from pedestrian data by relating distance to line-of-sight angle, then approximate them with speed-dependent ellipses enforced through a control-barrier function within a hybrid MPC planner. The learned zones are front-biased&#x2014;indicating humans prefer more space ahead&#x2014;and slightly tilted, reflecting overtaking behavior. While enforcing them as hard constraints improves safety, it can be overly conservative, since people typically tolerate reduced spacing in crowded settings. CASRL (<xref ref-type="bibr" rid="B357">Zhou et al., 2023</xref>) frames safety in navigation as a multi-task RL problem: goal reaching and collision avoidance tasks. It extends an off-policy RL algorithm (TD3) with separate critics for each task, while the actor is updated using a conflict-averse rule that maximizes the minimum improvement across tasks. This reduces performance loss when gradient updates disagree. However, the reported simulation gains fall short and may require further tuning.</p>
</sec>
<sec id="s2-5-2">
<label>2.5.2</label>
<title>Constrained RL</title>
<p>Constrained RL provides a natural framework for enforcing safety, as constraints take precedence over the reward objective when violated. For instance, <xref ref-type="bibr" rid="B236">Pfeiffer et al. (2018)</xref> introduce a safe RL navigation policy that defines a collision constraint, trained using constrained policy optimization (CPO) (<xref ref-type="bibr" rid="B1">Achiam et al., 2017</xref>), which maximizes reward while constraining the expected number of collisions. SoNIC (<xref ref-type="bibr" rid="B342">Yao et al., 2024</xref>) introduces a safety constraint derived from Adaptive Conformal Inference (ACI), which quantifies the uncertainty of predicted pedestrian trajectories. Similarly, <xref ref-type="bibr" rid="B361">Zhu et al. (2025)</xref> propose a confidence-weighted trajectory prediction model, where a Bayesian <inline-formula id="inf62">
<mml:math id="m62">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> parameter adapts the uncertainty region based on prediction errors. In their method, uncertainty is incorporated through a robust dynamical distance constraint that estimates time-to-collision, rather than relying on simple distance-based thresholds. However, both approaches employ trajectory predictors that neglect the robot&#x2019;s presence, resulting in predictions where pedestrians are assumed to move independently of the robot. This leads to overly conservative robot behavior.</p>
<p>In conclusion, safety-aware navigation improves reliability in learning-based systems through structured mechanisms, but further work is needed to balance safety with efficiency and ensure adaptability to diverse real-world scenarios.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Navigation model training</title>
<p>Training social navigation policies equips robots with safe, efficient, and socially aware navigation in human environments. This section outlines key training components (see <xref ref-type="fig" rid="F2">Figure 2</xref>), including the objective function, environments with static and dynamic obstacles, including realistic crowd simulation. Advanced strategies, such as pre-training, enhance training efficiency. We also examine human detection, tracking, prediction, and broader scene understanding and activity recognition, which are leveraged by navigation policies to improve performance. Finally, we cover evaluation methods for social navigation, including metrics and real-world experiments.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Illustration of the RL training loop, alternating between the Simulation Phase, where the navigation model (policy) interacts with the simulation environment, and the Learning Phase, where collected experience is used to improve the model through the RL algorithm.</p>
</caption>
<graphic xlink:href="frobt-12-1658643-g002.tif">
<alt-text content-type="machine-generated">Flowchart illustrating the reinforcement learning training loop divided into two phases. In the Simulation Phase, the policy interacts with the training environment, receiving state, action, and reward signals from the reward function. Collected experience is passed to the Learning Phase, where the RL algorithm updates the policy. A Training Enhancements module connects both phases, indicating optional methods that improve the learning process.</alt-text>
</graphic>
</fig>
<sec id="s3-1">
<label>3.1</label>
<title>Objective function</title>
<p>The objective or reward function in most reinforcement learning (RL) problems is typically formulated as <inline-formula id="inf63">
<mml:math id="m63">
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where the goal is to maximize the expected cumulative reward. This formulation typically incorporates a discount factor <inline-formula id="inf64">
<mml:math id="m64">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to prioritize earlier rewards over delayed ones. The reward function, denoted by <inline-formula id="inf65">
<mml:math id="m65">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, is often defined as a function of the current state and action but can also incorporate the subsequent state <inline-formula id="inf66">
<mml:math id="m66">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, to capture transition dynamics. In navigation tasks, the objective is often a weighted sum of reward components:<disp-formula id="equ1">
<mml:math id="m67">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf67">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are weights and <inline-formula id="inf68">
<mml:math id="m69">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are scalar and indicator functions. The section details possible reward components, but it&#x2019;s worth noting that combining multiple objectives can create local minima (<xref ref-type="bibr" rid="B81">Everett et al., 2021</xref>).</p>
<p>Many reward functions are <italic>sparse</italic>, providing feedback only at key milestones like reaching a goal. To improve learning, <italic>reward shaping</italic> introduces dense rewards, giving intermediate feedback at each timestep. While dense rewards speed up learning, they must be carefully designed to avoid suboptimal strategies.</p>
<sec id="s3-1-1">
<label>3.1.1</label>
<title>Goal reward</title>
<p>The reward function for reaching a goal state is the main component of any navigation task. It is often defined as an indicator function <inline-formula id="inf69">
<mml:math id="m70">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">tol</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, which returns 1 when the Euclidean distance between the robot&#x2019;s current position <inline-formula id="inf70">
<mml:math id="m71">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the goal position <inline-formula id="inf71">
<mml:math id="m72">
<mml:mrow>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is within a specified tolerance <inline-formula id="inf72">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">tol</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. A dense reward formulation for goal-reaching provides feedback based on progress, calculated as <inline-formula id="inf73">
<mml:math id="m74">
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>r</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>r</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, reflecting incremental advancement toward the goal (<xref ref-type="bibr" rid="B176">Long et al., 2018</xref>; <xref ref-type="bibr" rid="B292">Tan et al., 2020</xref>). Other approaches reward the agent for moving along grid cells that align with the global path (<xref ref-type="bibr" rid="B171">Liu Z. et al., 2023</xref>).</p>
</sec>
<sec id="s3-1-2">
<label>3.1.2</label>
<title>Collision-avoidance reward</title>
<p>The reward function for collision avoidance is often defined as an indicator function, <inline-formula id="inf74">
<mml:math id="m75">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">C</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which returns 1 when the robot&#x2019;s position <inline-formula id="inf75">
<mml:math id="m76">
<mml:mrow>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is within a collision set <inline-formula id="inf76">
<mml:math id="m77">
<mml:mrow>
<mml:mi mathvariant="script">C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, or when a collision is detected using other means. Alternative definitions penalize proximity to obstacles, such as <inline-formula id="inf77">
<mml:math id="m78">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">obs</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">obs</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">tol</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B42">Chen et al., 2017b</xref>) or <inline-formula id="inf78">
<mml:math id="m79">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">obs</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">tol</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">obs</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">tol</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B61">Cui et al., 2021</xref>) where <inline-formula id="inf79">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">obs</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the minimum distance to nearby obstacles and <inline-formula id="inf80">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">tol</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the tolerance distance. These formulations may apply different thresholds for static and dynamic obstacles, such as humans (<xref ref-type="bibr" rid="B61">Cui et al., 2021</xref>). Other approaches use step functions to gradually increase the penalty as the robot gets closer to obstacles, encouraging safer navigation.</p>
</sec>
<sec id="s3-1-3">
<label>3.1.3</label>
<title>Efficiency reward</title>
<p>To encourage efficient and timely navigation, reward functions often include terms that promote higher speeds. This may take the form of a gradual step function that provides a higher reward for increased velocity (<xref ref-type="bibr" rid="B153">Lee and Jeong, 2023</xref>) or a negative step-cost applied at each timestep to minimize time taken to reach the goal (<xref ref-type="bibr" rid="B312">Wang Y. et al., 2018</xref>; <xref ref-type="bibr" rid="B55">Choi et al., 2019</xref>).</p>
</sec>
<sec id="s3-1-4">
<label>3.1.4</label>
<title>Smoothness reward</title>
<p>For smooth trajectory generation, a negative reward proportional to the rotational velocity, <inline-formula id="inf81">
<mml:math id="m82">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is applied for differential drive robots (<xref ref-type="bibr" rid="B292">Tan et al., 2020</xref>; <xref ref-type="bibr" rid="B328">Xie and Dames, 2023</xref>). Additionally, <xref ref-type="bibr" rid="B124">Hoeller et al. (2021)</xref> penalizes lateral and backward velocities by adding a negative reward proportional to their squared magnitudes, encouraging smoother and more consistent forward motion.</p>
</sec>
<sec id="s3-1-5">
<label>3.1.5</label>
<title>Social reward</title>
<p>Social norms can be integrated into the reward function to promote behaviors like passing, crossing, and overtaking in socially appropriate ways (<xref ref-type="bibr" rid="B43">Chen et al., 2017c</xref>). This reward function is typically defined as a conditional function based on human parameters relative to the robot, including x-axis position, velocity, distance, relative heading angle, and heading angle difference. For instance, to promote overtaking from the left, the robot is rewarded when certain conditions are met: the goal distance exceeds 3, the human is positioned within <inline-formula id="inf82">
<mml:math id="m83">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf83">
<mml:math id="m84">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the robot&#x2019;s velocity surpasses the human&#x2019;s, and their heading angle difference is under <inline-formula id="inf84">
<mml:math id="m85">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s3-1-6">
<label>3.1.6</label>
<title>Geometric collision-avoidance reward</title>
<p>Model-based or geometric rewards using human position have enabled more robust navigation. For instance, <italic>DRL-VO</italic> (<xref ref-type="bibr" rid="B328">Xie and Dames, 2023</xref>) uses velocity obstacles (VOs) to model human motion, rewarding alignment with the optimal heading direction, where VOs are computed during training only. <xref ref-type="bibr" rid="B118">Han R. et al. (2022)</xref> incorporate VOs into both state and reward, with rewards based on joint VO area, velocity differences, and estimated minimum time to collision. <xref ref-type="bibr" rid="B360">Zhu et al. (2022)</xref> employ an oriented bounding capsule (OBC) model, where human velocity adds a buffer in front of the OBC, and the reward is the minimum distance to the OBC; OBC parameters are included in the robot&#x2019;s state for better learning. Additionally, <xref ref-type="bibr" rid="B266">Samsani and Muhammad (2021)</xref> define a danger zone (DZ) as an extended sector around humans, accounting for uncertainty in position and velocity predictions.</p>
</sec>
<sec id="s3-1-7">
<label>3.1.7</label>
<title>Human preference reward</title>
<p>Reinforcement Learning from Human Feedback (RLHF) provides a framework to simultaneously learn a policy and a reward function using human input (<xref ref-type="bibr" rid="B57">Christiano et al., 2017</xref>), with applications spanning various domains, including language models like GPT-3 (<xref ref-type="bibr" rid="B225">Ouyang et al., 2022</xref>). In social navigation, <xref ref-type="bibr" rid="B316">Wang R. et al. (2022)</xref> applies RLHF to learn a reward function based on pairwise human preferences over trajectory segments.</p>
</sec>
<sec id="s3-1-8">
<label>3.1.8</label>
<title>Human prediction reward</title>
<p>For planners that utilize human trajectory predictions, a negative reward is often used to discourage the robot from intruding into human-predicted zones (<xref ref-type="bibr" rid="B172">Liu S. et al., 2023</xref>). Additionally, a negative reward can be defined over prediction uncertainty, as in (<xref ref-type="bibr" rid="B77">Eiffert et al., 2020b</xref>), where the reward is the negative square root of the determinant of the covariance matrix, <inline-formula id="inf85">
<mml:math id="m86">
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula>, encouraging actions that lead to more predictable crowd behavior.</p>
</sec>
<sec id="s3-1-9">
<label>3.1.9</label>
<title>Exploration reward</title>
<p>Exploration rewards are designed to encourage the robot to explore a wide range of actions or states. Action-based exploration rewards promote action diversity by maximizing the policy&#x2019;s entropy (<xref ref-type="bibr" rid="B271">Schulman et al., 2017</xref>), while state-based exploration rewards encourage the robot to explore new areas. For instance, the intrinsic curiosity module (ICM) (<xref ref-type="bibr" rid="B230">Pathak et al., 2017</xref>), applied to navigation tasks (<xref ref-type="bibr" rid="B273">Shi H. et al., 2019</xref>; <xref ref-type="bibr" rid="B191">Martinez-Baselga et al., 2023</xref>) to reward the robot for discovering novel states, thereby enhancing its learning process.</p>
</sec>
<sec id="s3-1-10">
<label>3.1.10</label>
<title>Task-specific reward</title>
<p>Task-specific rewards are custom-designed to achieve the requirements of a particular navigation task. For example, in social navigation with a human companion, <xref ref-type="bibr" rid="B157">Li et al. (2018)</xref> define a distance-based reward that penalizes the robot for straying from its companion, encouraging it to stay close and coordinate its movement with the human partner.</p>
</sec>
<sec id="s3-1-11">
<label>3.1.11</label>
<title>Learning rewards from demonstrations</title>
<p>Inverse reinforcement learning (IRL) infers a reward function from expert demonstrations, either by using handcrafted state&#x2013;action features (<xref ref-type="bibr" rid="B223">Okal and Arras, 2016</xref>; <xref ref-type="bibr" rid="B142">Kim and Pineau, 2016</xref>) or by learning feature representations directly with neural networks (<xref ref-type="bibr" rid="B82">Fahad et al., 2018</xref>). For instance, <xref ref-type="bibr" rid="B307">Vasquez et al. (2014)</xref> learn a reward function expressed as a weighted combination of features that capture local crowd density, relative velocities and orientations of nearby pedestrians, the robot&#x2019;s own velocity, and social force interactions. More recently, methods like disturbance-based reward extrapolation (D-REX) (<xref ref-type="bibr" rid="B27">Brown et al., 2020</xref>) learn reward functions from suboptimal or unlabeled data. D-REX applies behavioral cloning, adds increasing <inline-formula id="inf86">
<mml:math id="m87">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-greedy noise, and trains the reward to favor less noisy trajectories, promoting desirable behavior, an approach effective for social navigation (<xref ref-type="bibr" rid="B69">De Heuvel et al., 2024</xref>). Similar techniques include T-REX (<xref ref-type="bibr" rid="B26">Brown et al., 2019</xref>) and SSRR (<xref ref-type="bibr" rid="B50">Chen et al., 2021</xref>).</p>
</sec>
<sec id="s3-1-12">
<label>3.1.12</label>
<title>Learning reward weights</title>
<p>Various techniques have been developed to automatically determine the optimal values of each objective weight <inline-formula id="inf87">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, eliminating the need for manual tuning. One such method is inverse reinforcement learning (IRL), which infers the weights <inline-formula id="inf88">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> by matching the robot&#x2019;s behavior to expert demonstrations (<xref ref-type="bibr" rid="B362">Ziebart et al., 2008</xref>). Additionally, AutoRL (<xref ref-type="bibr" rid="B53">Chiang et al., 2019</xref>; <xref ref-type="bibr" rid="B229">Parker-Holder et al., 2022</xref>) employs automated hyperparameter tuning to optimize the reward weights <inline-formula id="inf89">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> during training to enhance task-specific performance metrics.</p>
</sec>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>Training environment</title>
<p>This section reviews key components of training environments for social navigation, focusing on crowd data and physics-based simulators that replicate robot dynamics and sensory feedback to ensure realistic training conditions. Furthermore, crowd simulation libraries (see <xref ref-type="table" rid="T6">Table 6</xref>) provide controllable and realistic human behaviors that can be used to populate training environments and replicate crowd datasets.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Crowd simulation libraries.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Library</th>
<th align="left">Crowd behavior</th>
<th align="left">Language</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">RVO2 (<xref ref-type="bibr" rid="B305">Van Den Berg et al., 2025</xref>)</td>
<td align="left">ORCA</td>
<td align="left">Python (<xref ref-type="bibr" rid="B285">St&#xfc;vel, 2025</xref>) or C&#x2b;&#x2b;</td>
</tr>
<tr>
<td align="left">UMANS (<xref ref-type="bibr" rid="B306">van Toll et al., 2020</xref>)</td>
<td align="left">SFM, PLEdestrians, RVO ORCA, PowerLaw, Vision</td>
<td align="left">C&#x2b;&#x2b;</td>
</tr>
<tr>
<td align="left">PySocialForce (<xref ref-type="bibr" rid="B97">Gao, 2025</xref>)</td>
<td align="left">SFM with groups (<xref ref-type="bibr" rid="B210">Moussa&#xef;d et al., 2010</xref>)</td>
<td align="left">Python</td>
</tr>
<tr>
<td align="left">DeepSocialForce (<xref ref-type="bibr" rid="B149">Kreiss, 2021</xref>)</td>
<td align="left">Deep Social Force</td>
<td align="left">Python</td>
</tr>
<tr>
<td align="left">CROMOSIM (<xref ref-type="bibr" rid="B85">Faure, 2025</xref>)</td>
<td align="left">SFM, CA, Granular</td>
<td align="left">Python</td>
</tr>
<tr>
<td align="left">CrowdDynamics (<xref ref-type="bibr" rid="B106">Group, 2025</xref>)</td>
<td align="left">SFM with Path planning</td>
<td align="left">Python</td>
</tr>
<tr>
<td align="left">JuPedSim (<xref ref-type="bibr" rid="B74">Dynamics, 2025</xref>)</td>
<td align="left">SFM, Centrifugal, CFSM</td>
<td align="left">Python &#x26; C&#x2b;&#x2b;</td>
</tr>
<tr>
<td align="left">Mesa (<xref ref-type="bibr" rid="B192">Masad, and Kazil, 2015</xref>)</td>
<td align="left">ABM</td>
<td align="left">Python</td>
</tr>
<tr>
<td align="left">Agents.jl (<xref ref-type="bibr" rid="B66">Datseris et al., 2024</xref>)</td>
<td align="left">ABM</td>
<td align="left">Julia</td>
</tr>
<tr>
<td align="left">Vadere (<xref ref-type="bibr" rid="B145">Kleinmeier et al., 2019</xref>)</td>
<td align="left">CA, SFM, OSM (<xref ref-type="bibr" rid="B272">Seitz and K&#xf6;ster, 2012</xref>)</td>
<td align="left">Java</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3-2-1">
<label>3.2.1</label>
<title>Crowd data</title>
<p>Crowd datasets play a critical role in advancing data-driven approaches for both crowd behavior simulation and human trajectory prediction. They provide the necessary information to model realistic crowd interactions and dynamics, as detailed in the <xref ref-type="sec" rid="s11">Appendix</xref>. Additionally, these datasets support the training of human prediction methods, as explored in <xref ref-type="sec" rid="s3-3-2">Section 3.3.2</xref>. <xref ref-type="table" rid="T7">Table 7</xref> organizes these datasets by their sensory platforms, including stationary sensors, moving robots, and moving vehicles, each serving distinct purposes and applications. While long-term crowd tracking datasets such as the ATC dataset (<xref ref-type="bibr" rid="B28">Br&#x161;&#x10d;i&#x107; et al., 2013</xref>) exist, they lack the scale and diversity needed to support social navigation research.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Pedestrian datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dataset name</th>
<th align="left">Collection location</th>
<th align="left">Data volume</th>
<th align="left">Annotations</th>
<th align="left">Sensor platform</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">ETH (<xref ref-type="bibr" rid="B233">Pellegrini et al., 2009</xref>)</td>
<td align="left">ETH Zurich, and Hotel in Zurich</td>
<td align="left">25 min, 2 tracks</td>
<td align="left">Human position</td>
<td align="left">Stationary top-view camera</td>
</tr>
<tr>
<td align="left">UCY (<xref ref-type="bibr" rid="B156">Lerner et al., 2007</xref>)</td>
<td align="left">University campus and Zara store</td>
<td align="left">16 min, 6 tracks</td>
<td align="left">Human position</td>
<td align="left">Stationary top-view camera</td>
</tr>
<tr>
<td align="left">Central Station (<xref ref-type="bibr" rid="B355">Zhou et al., 2012</xref>)</td>
<td align="left">New York train station</td>
<td align="left">33 min</td>
<td align="left">Human position</td>
<td align="left">Stationary top-view camera</td>
</tr>
<tr>
<td align="left">Edinburgh (<xref ref-type="bibr" rid="B184">Majecka, 2009</xref>)</td>
<td align="left">Outdoor University campus</td>
<td align="left">92,000 trajectories</td>
<td align="left">Human position</td>
<td align="left">Stationary top-view camera</td>
</tr>
<tr>
<td align="left">Stanford Drone (<xref ref-type="bibr" rid="B256">Robicquet et al., 2016</xref>)</td>
<td align="left">Stanford University campus outdoor</td>
<td align="left">5 h, 60 tracks</td>
<td align="left">Bounding box</td>
<td align="left">Stationary drone top-view camera</td>
</tr>
<tr>
<td align="left">VIRAT (<xref ref-type="bibr" rid="B221">Oh et al., 2011</xref>)</td>
<td align="left">Indoor and outdoor</td>
<td align="left">25 h</td>
<td align="left">Bounding box</td>
<td align="left">Stationary surveillance camera</td>
</tr>
<tr>
<td align="left">Oxford Town (<xref ref-type="bibr" rid="B15">Benfold and Reid, 2011</xref>)</td>
<td align="left">Oxford town center</td>
<td align="left">5 min</td>
<td align="left">Bounding box</td>
<td align="left">Stationary surveillance camera</td>
</tr>
<tr>
<td align="left">ATC (<xref ref-type="bibr" rid="B28">Br&#x161;&#x10d;i&#x107; et al., 2013</xref>)</td>
<td align="left">Shopping mall</td>
<td align="left">92 days</td>
<td align="left">Human position</td>
<td align="left">Stationary top-mounted radar</td>
</tr>
<tr>
<td align="left">Ko-PER (<xref ref-type="bibr" rid="B284">Strigel et al., 2014</xref>)</td>
<td align="left">Crossing in Aschaffenburg, Germany</td>
<td align="left">6.5 min</td>
<td align="left">Human position</td>
<td align="left">Stationary top-view camera and Laser</td>
</tr>
<tr>
<td align="left">inD (<xref ref-type="bibr" rid="B23">Bock et al., 2020</xref>)</td>
<td align="left">Crossing in Aachen, Germany</td>
<td align="left">10 h</td>
<td align="left">Human position</td>
<td align="left">Stationary drone top-view camera</td>
</tr>
<tr>
<td align="left">CITR and DUT (<xref ref-type="bibr" rid="B339">Yang et al., 2019</xref>)</td>
<td align="left">University campus</td>
<td align="left">13 min, 66 tracks</td>
<td align="left">Human position</td>
<td align="left">Stationary drone top-view camera</td>
</tr>
<tr>
<td align="left">WILDTRACK (<xref ref-type="bibr" rid="B40">Chavdarova et al., 2018</xref>)</td>
<td align="left">ETH Zurich, Switzerland</td>
<td align="left">200 s</td>
<td align="left">Human position</td>
<td align="left">Stationary tilted camera</td>
</tr>
<tr>
<td align="left">STCrowd (<xref ref-type="bibr" rid="B59">Cong et al., 2022</xref>)</td>
<td align="left">Outdoor streets and walkways</td>
<td align="left">84 tracks</td>
<td align="left">Human position</td>
<td align="left">Stationary vehicle with RGB and 3D LiDAR</td>
</tr>
<tr>
<td align="left">THOR Dataset (<xref ref-type="bibr" rid="B263">Rudenko et al., 2020b</xref>)</td>
<td align="left">Indoor large room</td>
<td align="left">60 min</td>
<td align="left">Human position</td>
<td align="left">Stationary/moving robot with 3D LiDAR and tracking</td>
</tr>
<tr>
<td align="left">L-CAS (<xref ref-type="bibr" rid="B333">Yan et al., 2017</xref>)</td>
<td align="left">Indoor university building</td>
<td align="left">49 min</td>
<td align="left">Human position</td>
<td align="left">Stationary and moving robot with 3D LiDAR</td>
</tr>
<tr>
<td align="left">SCAND (<xref ref-type="bibr" rid="B137">Karnan et al., 2022</xref>)</td>
<td align="left">Indoor and outdoor</td>
<td align="left">8.7 h</td>
<td align="left">Social interaction</td>
<td align="left">Robot with RGB-D, and 3D LiDAR</td>
</tr>
<tr>
<td align="left">JRDB (<xref ref-type="bibr" rid="B190">Martin-Martin et al., 2021</xref>)</td>
<td align="left">Indoor and outdoor</td>
<td align="left">64 min</td>
<td align="left">Human position</td>
<td align="left">Robot with RGB-D, and 3D LiDAR</td>
</tr>
<tr>
<td align="left">FLOBOT (<xref ref-type="bibr" rid="B335">Yan et al., 2020</xref>)</td>
<td align="left">Airport and supermarket in Italy/France</td>
<td align="left">27 min</td>
<td align="left">Human position</td>
<td align="left">Autonomous robot with RGB-D, and 3D LiDAR</td>
</tr>
<tr>
<td align="left">NCLT (<xref ref-type="bibr" rid="B34">Carlevaris-Bianco et al., 2016</xref>)</td>
<td align="left">University campus indoor and outdoor</td>
<td align="left">35 h</td>
<td align="left">None</td>
<td align="left">Robot with RGB-D, and 3D LiDAR</td>
</tr>
<tr>
<td align="left">MuSoHu (<xref ref-type="bibr" rid="B219">Nguyen et al., 2023</xref>)</td>
<td align="left">Indoor and outdoor</td>
<td align="left">20 h</td>
<td align="left">Social interaction</td>
<td align="left">Walking human &#x2b; helmet with RGB-D and 3D LiDAR</td>
</tr>
<tr>
<td align="left">CrowdBot (<xref ref-type="bibr" rid="B226">Paez-Granados et al., 2021</xref>)</td>
<td align="left">3 streets in Lausanne, Switzerland</td>
<td align="left">200 min</td>
<td align="left">Human position</td>
<td align="left">Semi/Autonomous Robot with RGB-D and 3D LiDAR</td>
</tr>
<tr>
<td align="left">HuRoN (<xref ref-type="bibr" rid="B122">Hirose et al., 2023</xref>)</td>
<td align="left">5 office environments</td>
<td align="left">75 h</td>
<td align="left">Human position</td>
<td align="left">Autonomous robot with RGB-D and 2D LiDAR</td>
</tr>
<tr>
<td align="left">SiT (<xref ref-type="bibr" rid="B12">Bae et al., 2024</xref>)</td>
<td align="left">60 scenes, indoor and outdoor, Seoul</td>
<td align="left">470,000 frames</td>
<td align="left">Human position</td>
<td align="left">Robot with RGB-D, and 3D LiDAR</td>
</tr>
<tr>
<td align="left">KITTI (<xref ref-type="bibr" rid="B100">Geiger et al., 2012</xref>)</td>
<td align="left">Karlsruhe, Germany</td>
<td align="left">6 h</td>
<td align="left">Human position</td>
<td align="left">Moving vehicle with RGB, 3D LiDAR and more</td>
</tr>
<tr>
<td align="left">nuScences (<xref ref-type="bibr" rid="B30">Caesar et al., 2020</xref>)</td>
<td align="left">Boston, MA and Singapore</td>
<td align="left">5.5 h, 20s tracks</td>
<td align="left">Human position</td>
<td align="left">Moving vehicle with RGB, 3D LiDAR and more</td>
</tr>
<tr>
<td align="left">Waymo (<xref ref-type="bibr" rid="B79">Ettinger et al., 2021</xref>)</td>
<td align="left">San Francisco, Phoenix, and more</td>
<td align="left">570 h, 20s tracks</td>
<td align="left">Human position</td>
<td align="left">Moving vehicle with RGB, 3D LiDAR and more</td>
</tr>
<tr>
<td align="left">BDD100K (<xref ref-type="bibr" rid="B346">Yu et al., 2020</xref>)</td>
<td align="left">New York, Berkeley, and more</td>
<td align="left">1100 h, 40s tracks</td>
<td align="left">Human position</td>
<td align="left">Moving vehicle with RGB, 3D LiDAR and more</td>
</tr>
<tr>
<td align="left">A2D2 (<xref ref-type="bibr" rid="B101">Geyer et al., 2020</xref>)</td>
<td align="left">South of Germany</td>
<td align="left">40,000 frames</td>
<td align="left">Human position</td>
<td align="left">Moving vehicle with RGB, 3D LiDAR and more</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2-2">
<label>3.2.2</label>
<title>Simulation platform</title>
<p>Simulators provide a controlled virtual environment for developing and evaluating social navigation algorithms by modeling human-robot interactions and crowd behaviors. <xref ref-type="table" rid="T8">Table 8</xref> categorizes simulators based on key attributes, such as the supported sensor types, the human model ranging from simple cylindrical shapes to detailed 3D figures, supported crowd behaviors, evaluation metrics based on implementation specifics.</p>
<table-wrap id="T8" position="float">
<label>TABLE 8</label>
<caption>
<p>Simulation platforms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Simulator</th>
<th align="left">Description</th>
<th align="left">Human behavior</th>
<th align="left">Sensors</th>
<th align="left">Human model</th>
<th align="left">Maps</th>
<th align="left">Metrics</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Menge (<xref ref-type="bibr" rid="B62">Curtis et al., 2016</xref>)</td>
<td align="left">Supports ROS (<xref ref-type="bibr" rid="B10">Aroor et al., 2017</xref>)</td>
<td align="left">Path planning &#x2b; ORCA and SFM</td>
<td align="left">None</td>
<td align="left">Circle</td>
<td align="left">2D Maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">SEAN (<xref ref-type="bibr" rid="B302">Tsoi et al., 2022</xref>; <xref ref-type="bibr" rid="B300">Tsoi et al., 2020</xref>)</td>
<td align="left">Based on Unity game engine</td>
<td align="left">SFM and ORCA</td>
<td align="left">RGB-D</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">Navigation and Social</td>
</tr>
<tr>
<td align="left">SEAN-EP (<xref ref-type="bibr" rid="B301">Tsoi et al., 2021</xref>)</td>
<td align="left">Build on SEAN simulator For human feedback collection</td>
<td align="left">SFM and ORCA</td>
<td align="left">RGB-D</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">Navigation and Social</td>
</tr>
<tr>
<td align="left">UnrealCV (<xref ref-type="bibr" rid="B246">Qiu et al., 2017</xref>)</td>
<td align="left">Build on Unreal Engine 4</td>
<td align="left">None</td>
<td align="left">RGB</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">MORSE (<xref ref-type="bibr" rid="B75">Echeverria et al., 2011</xref>)</td>
<td align="left">Based on Blender engine</td>
<td align="left">None</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">Gazebo (<xref ref-type="bibr" rid="B146">Koenig and Howard, 2004</xref>)</td>
<td align="left">ROS compatible simulator</td>
<td align="left">None</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">Isaac Sim (<xref ref-type="bibr" rid="B185">Makoviychuk et al., 2021</xref>; <xref ref-type="bibr" rid="B204">Mittal et al., 2023</xref>)</td>
<td align="left">Based on PhysX engine</td>
<td align="left">None</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">3D Rigid human (Isaac GYM)</td>
<td align="left">None</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">Webots Sim (<xref ref-type="bibr" rid="B199">Michel, 2004</xref>)</td>
<td align="left">Based on OpenGL engine</td>
<td align="left">None</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">AI2-THOR (<xref ref-type="bibr" rid="B147">Kolve et al., 2017</xref>)</td>
<td align="left">Based on Unity Engine Only supports LoCoBot</td>
<td align="left">None</td>
<td align="left">RGB-D</td>
<td align="left">None</td>
<td align="left">3D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">Habitat 2.0 (<xref ref-type="bibr" rid="B288">Szot et al., 2021</xref>)</td>
<td align="left">Based Habitat Sim</td>
<td align="left">None</td>
<td align="left">RGB-D</td>
<td align="left">None</td>
<td align="left">3D Scans</td>
<td align="left">Navigation</td>
</tr>
<tr>
<td align="left">Habitat 3.0 (<xref ref-type="bibr" rid="B241">Puig et al., 2023</xref>)</td>
<td align="left">Based Habitat Sim</td>
<td align="left">Path Planning</td>
<td align="left">RGB-D</td>
<td align="left">3D humanoid</td>
<td align="left">3D Scans</td>
<td align="left">Navigation and Social</td>
</tr>
<tr>
<td align="left">HabiCrowd (<xref ref-type="bibr" rid="B310">Vuong et al., 2023</xref>)</td>
<td align="left">Based Habitat Sim</td>
<td align="left">UPL (<xref ref-type="bibr" rid="B136">Karamouzas et al., 2014</xref>)</td>
<td align="left">RGB-D</td>
<td align="left">3D humanoid</td>
<td align="left">3D Scans</td>
<td align="left">Navigation and Social</td>
</tr>
<tr>
<td align="left">SAPIEN (<xref ref-type="bibr" rid="B327">Xiang et al., 2020</xref>)</td>
<td align="left">Based on PhysX engine</td>
<td align="left">None</td>
<td align="left">RGB-D</td>
<td align="left">None</td>
<td align="left">3D Scans</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">CrowdBot Sim (<xref ref-type="bibr" rid="B107">Grzeskowiak et al., 2021</xref>)</td>
<td align="left">Based on Unity Engine</td>
<td align="left">UMANS (<xref ref-type="bibr" rid="B306">van Toll et al., 2020</xref>)</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">Navigation and Social</td>
</tr>
<tr>
<td align="left">iGibson (<xref ref-type="bibr" rid="B159">Li et al., 2021</xref>)</td>
<td align="left">Based on PyBullet</td>
<td align="left">ORCA</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">3D Rigid human</td>
<td align="left">3D Scans</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">CrowdNav (<xref ref-type="bibr" rid="B47">Chen et al., 2019b</xref>)</td>
<td align="left">2D openspace simulator</td>
<td align="left">SFM and ORCA</td>
<td align="left">None</td>
<td align="left">Circle</td>
<td align="left">None</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">Arena-Rosnav (<xref ref-type="bibr" rid="B138">K&#xe4;stner et al., 2021</xref>)</td>
<td align="left">Uses Gazebo, Unity and Flatland</td>
<td align="left">SFM and ORCA</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">Circle or 3D humanoid</td>
<td align="left">2D and 3D maps</td>
<td align="left">Navigation and Social</td>
</tr>
<tr>
<td align="left">nav-gym (<xref ref-type="bibr" rid="B154">Lee et al., 2023</xref>)</td>
<td align="left">Use CrowdNav with Flatland</td>
<td align="left">ORCA</td>
<td align="left">2D LiDAR</td>
<td align="left">Circle or 3D humanoid</td>
<td align="left">2D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">gym-collision<break/>-avoidance (<xref ref-type="bibr" rid="B80">Everett et al., 2018</xref>)</td>
<td align="left">Supports multiple robots and humans</td>
<td align="left">ORCA</td>
<td align="left">None</td>
<td align="left">Circle</td>
<td align="left">None</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">navrep (<xref ref-type="bibr" rid="B73">Dugas et al., 2021</xref>)</td>
<td align="left">Multiple social simulators</td>
<td align="left">Global planner &#x2b; ORCA</td>
<td align="left">2D LiDAR</td>
<td align="left">Circle or human with moving legs</td>
<td align="left">2D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">gym ped sim (<xref ref-type="bibr" rid="B290">Tai et al., 2018</xref>)</td>
<td align="left">A plugin for Gazebo</td>
<td align="left">SFM</td>
<td align="left">RGB-D and 3D LiDAR</td>
<td align="left">3D humanoid</td>
<td align="left">3D maps</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">Social Gym (<xref ref-type="bibr" rid="B282">Sprague et al., 2023</xref>)</td>
<td align="left">A multi-agent 2D simulator</td>
<td align="left">RL planner</td>
<td align="left">None</td>
<td align="left">Circle</td>
<td align="left">2D maps</td>
<td align="left">Navigation and Social</td>
</tr>
<tr>
<td align="left">RDS Sim (<xref ref-type="bibr" rid="B105">Gonon et al., 2021</xref>)</td>
<td align="left">RDS planner evaluation code</td>
<td align="left">ORCA</td>
<td align="left">None</td>
<td align="left">Circle</td>
<td align="left">None</td>
<td align="left">Navigation and Social</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Many simulators share a common emphasis on creating realistic environments. For instance, Habitat (<xref ref-type="bibr" rid="B288">Szot et al., 2021</xref>) and Gibson (<xref ref-type="bibr" rid="B159">Li et al., 2021</xref>), widely used in embodied AI research, render highly detailed indoor spaces using real 3D scans. However, these environments are typically limited to smaller areas like apartments or offices, making them less suitable for large-scale crowd simulations. Additionally, several simulators, such as Isaac Sim (<xref ref-type="bibr" rid="B185">Makoviychuk et al., 2021</xref>), prioritize achieving high FPS, which is crucial for training performance. Simulators also vary in the complexity of crowd behaviors, with some supporting basic movement patterns and others providing sophisticated, behavior-rich models that more accurately capture crowd dynamics, such as NavRep (<xref ref-type="bibr" rid="B73">Dugas et al., 2021</xref>). However, current simulators remain limited, as an efficient RL-supported crowd simulation with diverse scenarios is still missing, which we aim to address with our benchmark.</p>
</sec>
</sec>
<sec id="s3-3">
<label>3.3</label>
<title>Human detection, tracking, and prediction</title>
<p>In social navigation, a robot often relies on human detection, tracking, or prediction for better social awareness and generalizability. <italic>Human Detection</italic> provides the robot-centric human positions, which are essential for position-based planners (see <xref ref-type="sec" rid="s2-2">Section 2.2</xref>). <italic>Human Tracking</italic> estimates human positions and velocities over time, supporting planners requiring human speeds or trajectories. <italic>Human Prediction</italic> utilizes tracking data to forecast future human movements, which are utilized by prediction-based planners (see <xref ref-type="sec" rid="s2-4">Section 2.4</xref>).</p>
<sec id="s3-3-1">
<label>3.3.1</label>
<title>Human detection and tracking</title>
<p>Human detection methods are typically tailored to specific sensors, including 2D LiDAR, 3D LiDAR, RGB, and RGB-D sensors. Tracking enhances detection by assigning unique identifiers to individuals and addressing challenges such as sensor occlusions, which is essential for reliable multi-object tracking (MOT). While human detection provides the robot-centric positions of each detected person, tracking maintains a history of these positions over time, enabling the estimation of their velocities.</p>
<sec id="s3-3-1-1">
<label>3.3.1.1</label>
<title>Human detection</title>
<p>RGB-based human detection leverages general object detection techniques, which can be broadly categorized into classical and deep learning approaches. Classical methods such as histogram of oriented gradients (HOG) (<xref ref-type="bibr" rid="B65">Dalal and Triggs, 2005</xref>) and deformable part model (DPM) (<xref ref-type="bibr" rid="B87">Felzenszwalb et al., 2008</xref>), often struggle with accuracy and robustness in complex or dynamic environments. Deep learning-based methods, on the other hand, are divided into one-stage and two-stage approaches. Two-stage or <italic>coarse-to-fine</italic> methods, like Faster R-CNN (<xref ref-type="bibr" rid="B251">Ren et al., 2016</xref>) and FPN (<xref ref-type="bibr" rid="B163">Lin et al., 2017</xref>), typically offer higher accuracy by refining proposals. While one-stage detectors, such as YOLO (<xref ref-type="bibr" rid="B249">Redmon, 2016</xref>), SSD (<xref ref-type="bibr" rid="B166">Liu et al., 2016</xref>), and DETR (<xref ref-type="bibr" rid="B33">Carion et al., 2020</xref>), prioritize speed, making them ideal for real-time applications in social navigation. For further details, refer to <xref ref-type="bibr" rid="B363">Zou et al. (2023)</xref>. The output of RGB-based object detection provides a bounding box in the image plane, which requires conversion to robot-centered coordinates for accurate spatial positioning. To estimate 3D pose parameters from 2D detections, some methods, such as Multi-fusion (<xref ref-type="bibr" rid="B330">Xu and Chen, 2018</xref>) and ROI-10D (<xref ref-type="bibr" rid="B186">Manhardt et al., 2019</xref>), incorporate depth estimation modules to approximate distance. Meanwhile, techniques like Deep3DBox (<xref ref-type="bibr" rid="B209">Mousavian et al., 2017</xref>), MonoGRnet (<xref ref-type="bibr" rid="B244">Qin et al., 2019</xref>), and <xref ref-type="bibr" rid="B126">Hu et al. (2019)</xref> apply geometric reasoning techniques for 3D localization based on 2D information.</p>
<p>Early methods for 2D LiDAR-based human detection relied on hand-crafted features, identifying humans by detecting both legs within a segment (<xref ref-type="bibr" rid="B11">Arras et al., 2007</xref>) or by tracking individual legs over time (<xref ref-type="bibr" rid="B155">Leigh et al., 2015</xref>). The first deep learning-based detector, <italic>DROW</italic> (<xref ref-type="bibr" rid="B18">Beyer et al., 2016</xref>), was subsequently enhanced by incorporating temporal information to improve tracking consistency (<xref ref-type="bibr" rid="B19">Beyer et al., 2018</xref>). Building upon <italic>DROW</italic>, <italic>DR-SPAAM</italic> (<xref ref-type="bibr" rid="B134">Jia et al., 2020</xref>) introduced faster processing capabilities for handling long-term temporal data. Additionally, <xref ref-type="bibr" rid="B70">Dequaire et al. (2018)</xref> employed an occupancy grid-based approach combined with an RNN to capture temporal patterns effectively. Current 3D LiDAR detection approaches are categorized into Bird&#x2019;s Eye View (BEV) methods, point-based methods, voxel-based methods, multi-view methods, and range-view-based methods. <italic>BEV</italic> methods provide fast, top-down 2D projections of the environment, making them popular for quick processing tasks in robotics. Examples include PIXOR (<xref ref-type="bibr" rid="B337">Yang et al., 2018a</xref>) and HDNet (<xref ref-type="bibr" rid="B338">Yang et al., 2018b</xref>). However, they often miss critical vertical details essential for detecting objects like pedestrians. <italic>Point-based</italic> methods directly process raw point cloud data, offering higher accuracy. Notable examples are PointNet&#x2b;&#x2b; (<xref ref-type="bibr" rid="B242">Qi et al., 2017</xref>) and PointRCNN (<xref ref-type="bibr" rid="B274">Shi S. et al., 2019</xref>). However, these methods are computationally intensive and less suitable for real-time applications. <italic>Voxel-based</italic> methods transform point clouds into 3D voxel grids, effectively balancing accuracy and computational efficiency by reducing processing loads while preserving essential details. Notable examples include VoxelNet (<xref ref-type="bibr" rid="B354">Zhou and Tuzel, 2018</xref>) and SECOND (<xref ref-type="bibr" rid="B334">Yan et al., 2018</xref>). <italic>Multi-view</italic> methods, such as MV3D (<xref ref-type="bibr" rid="B44">Chen X. et al., 2017</xref>) and SE-SSD (<xref ref-type="bibr" rid="B353">Zheng et al., 2021</xref>), combine multiple point cloud representations to leverage their respective advantages and enhance detection performance. <italic>Range-view-based</italic> methods convert LiDAR data into 2D range images, preserving vertical details and achieving high processing speeds, making them well-suited for applications like social navigation. Approaches include RangeNet&#x2b;&#x2b; (<xref ref-type="bibr" rid="B200">Milioto et al., 2019</xref>) and RSN (<xref ref-type="bibr" rid="B287">Sun et al., 2021</xref>). RGB-D-based human detection combines RGB data with depth information, which can also be acquired from a 3D LiDAR for sensor fusion. Techniques like PointPainting (<xref ref-type="bibr" rid="B308">Vora et al., 2020</xref>) fuse RGB semantic data onto LiDAR points, while PointNet (<xref ref-type="bibr" rid="B243">Qi et al., 2018</xref>) leverage 3D bounding frustums, focusing detection within the RGB-D space. For further details, refer to <xref ref-type="bibr" rid="B187">Mao J. et al. (2023)</xref>.</p>
</sec>
<sec id="s3-3-1-2">
<label>3.3.1.2</label>
<title>Human tracking</title>
<p>Human tracking involves identifying detected objects, assigning each object a unique ID, and continuously updating their location through state estimation filters, even during brief sensor occlusions. This section centers on the Tracking-by-Detection framework, which performs detection before tracking, as other tracking frameworks are less common for human tracking. Trackers vary by association metrics and tracking dimensionality. In general, MOT relies on motion prediction techniques such as Kalman filters, particle filters, or multi-hypothesis tracking (MHT) (<xref ref-type="bibr" rid="B345">Yoon et al., 2018</xref>), combined with application-specific association metrics (<xref ref-type="bibr" rid="B248">Rakai et al., 2022</xref>). For vision-based MOT, popular methods include DEEPSort (<xref ref-type="bibr" rid="B325">Wojke et al., 2017</xref>) which integrates deep association metrics, ByteTrack (<xref ref-type="bibr" rid="B351">Zhang et al., 2022</xref>) which relies on hierarchical association for accurate initial detection and faster performance, and other methods (<xref ref-type="bibr" rid="B332">Xu Y. et al., 2019</xref>). For 3D MOT, approaches like AB3DMOT (<xref ref-type="bibr" rid="B322">Weng et al., 2020</xref>), which utilizes 3D bounding boxes and Kalman filtering, and other approaches like SimpleTrack (<xref ref-type="bibr" rid="B228">Pang et al., 2022</xref>) and CAMO-MOT (<xref ref-type="bibr" rid="B320">Wang L. et al., 2023</xref>) which enhance tracking accuracy and efficiency. Fusion-based MOT combines 2D and 3D detections from multiple sensors to enhance tracking robustness. EagerMOT (<xref ref-type="bibr" rid="B143">Kim et al., 2021</xref>) fuses information from multiple detectors, while DeepfusionMOT (<xref ref-type="bibr" rid="B317">Wang X. et al., 2022</xref>) applies deep learning-based association for enhanced consistency. For further details, refer to <xref ref-type="bibr" rid="B234">Peng et al. (2024)</xref>.</p>
</sec>
</sec>
<sec id="s3-3-2">
<label>3.3.2</label>
<title>Human trajectory prediction</title>
<p>Predicting human trajectories is critical for effective social navigation. Traditionally relying on knowledge-based methods, the field has shifted towards learning-based approaches, which consistently outperform traditional methods on metrics such as average displacement error (ADE) (<xref ref-type="bibr" rid="B233">Pellegrini et al., 2009</xref>) and final displacement error (FDE) (<xref ref-type="bibr" rid="B2">Alahi et al., 2016</xref>). Learning-based methods leverage crowd datasets (see <xref ref-type="sec" rid="s3-2-1">Section 3.2.1</xref>) and typically employ CNN, LSTM, or GAN architectures (<xref ref-type="bibr" rid="B148">Korbmacher and Tordeux, 2022</xref>).</p>
<sec id="s3-3-2-1">
<label>3.3.2.1</label>
<title>CNN-based predictors</title>
<p>CNNs, initially designed for spatial tasks, have been adapted to sequential pedestrian prediction by representing trajectories spatially. Early approaches such as Behavior-CNN (<xref ref-type="bibr" rid="B344">Yi et al., 2016</xref>) encode pedestrian trajectories into displacement volumes processed by CNN layers. More recent models, such as Social-STGCNN (<xref ref-type="bibr" rid="B205">Mohamed et al., 2020</xref>), incorporate graph convolutions to effectively model pedestrian interactions, while scene context integration further enhances predictions (<xref ref-type="bibr" rid="B253">Ridel et al., 2020</xref>). Overall, CNNs efficiently process data in parallel but typically require reprocessing the full input history for each prediction, limiting their efficiency in real-time navigation.</p>
</sec>
<sec id="s3-3-2-2">
<label>3.3.2.2</label>
<title>LSTM-based predictors</title>
<p>LSTM networks excel at capturing temporal dependencies in sequential data. Social-LSTM (<xref ref-type="bibr" rid="B2">Alahi et al., 2016</xref>) introduced social pooling to account for pedestrian interactions during prediction. Enhancements include integrating environmental context via semantic information (<xref ref-type="bibr" rid="B165">Lisotto et al., 2019</xref>) and employing attention mechanisms (<xref ref-type="bibr" rid="B88">Fernando et al., 2018</xref>). Graph-based methods like STGAT (<xref ref-type="bibr" rid="B127">Huang et al., 2019</xref>) further improve interaction modeling. Transformers have recently emerged as powerful alternatives, better capturing complex interactions and limited sensing scenarios (<xref ref-type="bibr" rid="B128">Huang et al., 2021</xref>). In contrast, LSTMs, despite slower batch processing, efficiently leverage hidden states for incremental, real-time predictions, making them ideal for social navigation.</p>
</sec>
<sec id="s3-3-2-3">
<label>3.3.2.3</label>
<title>GAN-based predictors</title>
<p>GAN-based models generate diverse and realistic trajectories, addressing human behavior&#x2019;s multi-modality. Influential methods include Social-GAN (<xref ref-type="bibr" rid="B110">Gupta et al., 2018</xref>), which combines LSTMs with GAN frameworks, and SoPhie (<xref ref-type="bibr" rid="B265">Sadeghian et al., 2019</xref>), which integrates social and physical context through attention modules. Recent advancements like probabilistic crowd GAN (PCGAN) (<xref ref-type="bibr" rid="B77">Eiffert et al., 2020b</xref>) and diffusion-based models (<xref ref-type="bibr" rid="B108">Gu et al., 2022</xref>; <xref ref-type="bibr" rid="B188">Mao W. et al., 2023</xref>) further enhance multi-modal, safety-compliant predictions. Despite the computational demand, GANs&#x2019; diverse trajectory predictions significantly contribute to robust and safe decision-making in social navigation scenarios.</p>
</sec>
</sec>
</sec>
<sec id="s3-4">
<label>3.4</label>
<title>Scene understanding and activity recognition</title>
<p>Scene understanding and activity recognition are perception modules that provide information beyond human detection and trajectory prediction. Scene understanding includes object detection, pose estimation, semantic segmentation, saliency prediction, affordance prediction, and captioning (<xref ref-type="bibr" rid="B218">Naseer et al., 2018</xref>).</p>
<p>Object detection and pose estimation, detailed in <xref ref-type="sec" rid="s3-3">Section 3.3</xref> for humans, can be generalized to other classes for broader scene understanding. Beyond object detection, 2D and 3D semantic segmentation assign semantic labels to pixels or points in images and LiDAR scans, producing detailed maps of the environment (<xref ref-type="bibr" rid="B144">Kirillov et al., 2023</xref>; <xref ref-type="bibr" rid="B35">Cen et al., 2023</xref>) with applications to navigation (<xref ref-type="bibr" rid="B261">Roth et al., 2024</xref>). Affordance prediction further interprets the scene by modeling possible interactions; for navigation, this is useful for identifying robot-traversable areas (<xref ref-type="bibr" rid="B347">Yuan et al., 2024</xref>). Saliency prediction models human visual attention by estimating focus regions in a scene (<xref ref-type="bibr" rid="B178">Lou et al., 2022</xref>), allowing vision models to ignore irrelevant input and prioritize informative areas. Finally, 3D dense captioning methods, such as Vote2Cap-DETR (<xref ref-type="bibr" rid="B51">Chen et al., 2023</xref>), extend scene classification or 2D captioning by generating multiple localized captions, offering richer scene descriptions for context-aware navigation.</p>
<p>In parallel, activity recognition interprets dynamic human behaviors at both the individual and group levels. At the individual level, this involves human action classification (<xref ref-type="bibr" rid="B102">Girdhar et al., 2017</xref>), while at the group level it includes group activity classification (<xref ref-type="bibr" rid="B54">Choi et al., 2009</xref>) often supported by group detection methods (<xref ref-type="bibr" rid="B314">Wang Q. et al., 2018</xref>; <xref ref-type="bibr" rid="B160">Li et al., 2022</xref>). More recently, LLM-based classifiers have been introduced for activity recognition (<xref ref-type="bibr" rid="B247">Qu et al., 2024</xref>; <xref ref-type="bibr" rid="B174">Liu et al., 2025</xref>). Current navigation approaches primarily use activity recognition to estimate proxemics (<xref ref-type="bibr" rid="B38">Charalampous et al., 2016</xref>; <xref ref-type="bibr" rid="B215">Narayanan et al., 2020</xref>), though its potential for richer context-aware decision-making remains unexplored.</p>
<p>Vision-language models (VLMs) (<xref ref-type="bibr" rid="B173">Liu H. et al., 2023</xref>) are large multimodal models with broad capabilities, including object recognition, reasoning, and contextual understanding. By jointly leveraging visual and textual inputs, they provide a natural bridge between scene understanding, activity recognition, and navigation guidance. Despite this potential, their use in social navigation remains limited, with only a few recent methods exploring VLM-based decision making (<xref ref-type="bibr" rid="B281">Song et al., 2024</xref>; <xref ref-type="bibr" rid="B211">Munje et al., 2025</xref>).</p>
</sec>
<sec id="s3-5">
<label>3.5</label>
<title>Training enhancement techniques</title>
<p>Efficient training is essential for robust social navigation policies, since large-scale RL training is often limited by computational resources. While extensive training, such as training a DD-PPO policy for 2 billion steps <xref ref-type="bibr" rid="B323">Wijmans et al. (2019)</xref>, can boost performance, more efficient approaches exist. Task-specific techniques, such as leveraging problem symmetries by flipping path topologies (<xref ref-type="bibr" rid="B43">Chen et al., 2017c</xref>) can improve exploration. This section highlights general, task-agnostic methods for enhancing training efficiency and performance.</p>
<sec id="s3-5-1">
<label>3.5.1</label>
<title>Pre-training techniques</title>
<p>Pre-training techniques, such as behavioral cloning from demonstrations (<xref ref-type="bibr" rid="B236">Pfeiffer et al., 2018</xref>; <xref ref-type="bibr" rid="B47">Chen C. et al., 2019</xref>), accelerate training by providing basic navigation skills and reducing RL exploration. Self-supervised methods, like VAEs with reconstruction loss (<xref ref-type="bibr" rid="B73">Dugas et al., 2021</xref>; <xref ref-type="bibr" rid="B124">Hoeller et al., 2021</xref>), improve state representation, while transfer learning from pretrained CNNs enhances RGB input processing (<xref ref-type="bibr" rid="B125">Hong et al., 2021</xref>). Policy transfer from existing models is also used (<xref ref-type="bibr" rid="B323">Wijmans et al., 2019</xref>). These approaches improve training efficiency, convergence, and generalization.</p>
</sec>
<sec id="s3-5-2">
<label>3.5.2</label>
<title>Auxiliary tasks</title>
<p>Auxiliary tasks are additional tasks or objectives incorporated during training to support learning the main task. This offers better training signal and model performance. Auxiliary tasks have been shown to improve navigation performance by training models to predict features such as depth, loop closures (<xref ref-type="bibr" rid="B202">Mirowski et al., 2016</xref>), and location estimation (<xref ref-type="bibr" rid="B297">Tongloy et al., 2017</xref>). Additional tasks include predicting immediate reward prediction and learning to control specific regions in the input image (<xref ref-type="bibr" rid="B131">Jaderberg et al., 2016</xref>) or predicting image segmentation (<xref ref-type="bibr" rid="B151">Kulh&#xe1;nek et al., 2019</xref>). In social navigation, auxiliary tasks are used to improve understanding of social dynamics. For instance, <italic>Proximity-Aware</italic> (<xref ref-type="bibr" rid="B32">Cancelli et al., 2023</xref>) incorporates tasks to estimate the distance and direction of surrounding humans, while <italic>Falcon</italic> (<xref ref-type="bibr" rid="B104">Gong et al., 2024</xref>) incorporates tasks for predicting the number of nearby humans, tracking their locations, and estimating their future trajectories. These tasks enable the model to acquire valuable insights into the environment&#x2019;s social dynamics, leading to more efficient and informed planning.</p>
</sec>
<sec id="s3-5-3">
<label>3.5.3</label>
<title>Curriculum learning</title>
<p>Curriculum learning gradually increases task difficulty during training, aiding convergence in challenging social navigation tasks. In RL, this process involves three steps: task generation, sequencing, and transfer learning (<xref ref-type="bibr" rid="B217">Narvekar et al., 2020</xref>). <italic>Task generation</italic> creates scenarios of varying difficulty by adjusting obstacles, goal distances, or map complexity, using parameter sampling or grid search. <italic>Sequencing</italic> organizes tasks by increasing difficulty, either at a fixed rate or adaptively based on agent performance, and may involve modifying reward functions or start/goal distributions (<xref ref-type="bibr" rid="B254">Riedmiller et al., 2018</xref>; <xref ref-type="bibr" rid="B92">Florensa et al., 2018</xref>), optimization strategies (<xref ref-type="bibr" rid="B194">Matiisen et al., 2019</xref>), Curriculum MDPs (<xref ref-type="bibr" rid="B216">Narvekar et al., 2017</xref>), or human feedback (<xref ref-type="bibr" rid="B16">Bengio et al., 2009</xref>). <italic>Transfer learning</italic> adapts agents when intermediate tasks differ in state/action spaces, rewards, or dynamics, such as transitioning from precise states to noisy sensors, or from indoor to outdoor navigation. This combination allows agents to efficiently learn complex social navigation skills.</p>
</sec>
<sec id="s3-5-4">
<label>3.5.4</label>
<title>Teacher-student framework</title>
<p>The teacher-student framework enables a teacher model, often trained with privileged information, to guide a student via real-time feedback, reward shaping, or action labels. Knowledge transfer is achieved through policy distillation (<xref ref-type="bibr" rid="B264">Rusu et al., 2015</xref>), using labeled paths or actions from the teacher, student, or both (<xref ref-type="bibr" rid="B63">Czarnecki et al., 2018</xref>), allowing the student to imitate and refine its navigation policy, which can later be fine-tuned with RL. Teachers may also provide reward signals to enhance exploration (<xref ref-type="bibr" rid="B64">Czarnecki et al., 2019</xref>) and corrective action feedback (<xref ref-type="bibr" rid="B259">Ross et al., 2011</xref>). Model-based teachers like MPC are also used (<xref ref-type="bibr" rid="B179">Lowrey et al., 2018</xref>). Asymmetric actor-critic methods allow the critic to use privileged information to guide the actor (<xref ref-type="bibr" rid="B237">Pinto et al., 2017</xref>). In teacher-student curriculum learning, teachers assign progressively harder tasks and are rewarded for student improvement (<xref ref-type="bibr" rid="B194">Matiisen et al., 2019</xref>), while multi-teacher approaches combine skills from specialized teachers (<xref ref-type="bibr" rid="B264">Rusu et al., 2015</xref>). For social navigation, non-optimal teachers (e.g., PID planners) can be combined with RL, accelerating training by switching to the higher Q-value source (<xref ref-type="bibr" rid="B329">Xie et al., 2018</xref>).</p>
</sec>
<sec id="s3-5-5">
<label>3.5.5</label>
<title>Sim-to-real</title>
<p>Sim-to-real transfer for navigation tackles the challenge of adapting a simulation-trained policy to perform reliably in real-world environments. Achieving sim-to-real transfer requires a highly realistic simulator (refer to <xref ref-type="sec" rid="s3-2-2">Section 3.2.2</xref>) and the implementation of techniques like domain randomization and domain adaptation. These techniques operate at different levels: scenario-level randomization and adaptation (see <xref ref-type="sec" rid="s11">Appendix</xref> for details) modify various aspects of the simulated environment, while sensor-level noise enables the policy to handle discrepancies in real-world sensor data. <italic>Domain adaptation</italic> adjusts simulation-trained models to real-world domains. For RGB data, this uses real-world samples and methods like discrepancy minimization, adversarial alignment, or reconstruction methods for feature alignment (<xref ref-type="bibr" rid="B311">Wang and Deng, 2018</xref>). For depth sensors, techniques such as depth completion and refinement address real-world limitations, improving consistency with simulated data (<xref ref-type="bibr" rid="B141">Khan et al., 2022</xref>). <italic>Domain randomization</italic> narrows the sim-to-real gap by introducing simulated variability, allowing policies to generalize to real-world conditions (<xref ref-type="bibr" rid="B296">Tobin et al., 2017</xref>). For RGB inputs, this includes varying visual features to simulate lighting and color changes (<xref ref-type="bibr" rid="B8">Anderson et al., 2021</xref>); for depth sensors, it involves adding noise, occlusions, warping, and quantization (<xref ref-type="bibr" rid="B212">Muratore et al., 2022</xref>; <xref ref-type="bibr" rid="B293">Thalhammer et al., 2019</xref>). Active domain randomization further improves robustness by focusing on model-effecting variations (<xref ref-type="bibr" rid="B198">Mehta et al., 2020</xref>; <xref ref-type="bibr" rid="B348">Zakharov et al., 2019</xref>).</p>
</sec>
</sec>
<sec id="s3-6">
<label>3.6</label>
<title>Navigation model evaluation</title>
<p>Evaluating social navigation policies requires a robust approach to ensure reliable and safe robot operation in human environments. This section covers policy evaluation by outlining real-world experiments that validate a robot&#x2019;s capabilities in realistic, dynamic settings and by presenting metrics that offer structured, quantifiable insights into both navigation performance and social compliance. For a more comprehensive overview of social navigation evaluation, see <xref ref-type="bibr" rid="B95">Francis et al. (2023)</xref> and <xref ref-type="bibr" rid="B98">Gao and Huang (2022)</xref>.</p>
<sec id="s3-6-1">
<label>3.6.1</label>
<title>Real-world experiments</title>
<p>Evaluating social navigation policies in real-world settings is crucial for assessing their robustness, adaptability, and social acceptability. Experiments typically fall into three categories: experimental demonstrations, lab studies, and field studies (<xref ref-type="bibr" rid="B197">Mavrogiannis et al., 2023</xref>). Experimental demonstrations offer proof-of-concept with limited reproducibility (<xref ref-type="bibr" rid="B42">Chen et al., 2017b</xref>; <xref ref-type="bibr" rid="B47">Chen C. et al., 2019</xref>), while lab studies provide structured, repeatable tests in controlled environments with systematic reporting (<xref ref-type="bibr" rid="B299">Tsai and Oh, 2020</xref>; <xref ref-type="bibr" rid="B196">Mavrogiannis et al., 2019</xref>). Field studies are the most comprehensive, deploying robots in public spaces among uninstructed pedestrians (<xref ref-type="bibr" rid="B139">Kato et al., 2015</xref>; <xref ref-type="bibr" rid="B142">Kim and Pineau, 2016</xref>). Real-world evaluations combine quantitative metrics with qualitative observations, such as participant feedback or questionnaires, to assess social adaptability and compliance (<xref ref-type="bibr" rid="B238">Pirk et al., 2022</xref>).</p>
</sec>
<sec id="s3-6-2">
<label>3.6.2</label>
<title>Metrics</title>
<p>Navigation and social navigation metrics provide a structured framework to assess robot performance in crowded environments. Traditional navigation metrics assess robots&#x2019; fundamental abilities such as reaching targets and avoiding obstacles, while social navigation metrics focus on interactions with humans, including maintaining personal space and minimizing disruptions to bystanders. Together these metrics, as detailed in <xref ref-type="table" rid="T9">Table 9</xref>, guide the development of navigation systems that achieve task objectives efficiently while adhering to socially appropriate behaviors, promoting safer and widely accepted robot deployments.</p>
<table-wrap id="T9" position="float">
<label>TABLE 9</label>
<caption>
<p>Navigation and social navigation metrics.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Metric</th>
<th align="left">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Success</td>
<td align="left">Goal-reaching success rate</td>
</tr>
<tr>
<td align="left">Timeout</td>
<td align="left">Exceeded time limit runs</td>
</tr>
<tr>
<td align="left">Path Length</td>
<td align="left">Total path distance</td>
</tr>
<tr>
<td align="left">Run Time</td>
<td align="left">Completion time</td>
</tr>
<tr>
<td align="left">SPL (<xref ref-type="bibr" rid="B7">Anderson et al., 2018b</xref>)</td>
<td align="left">Path efficiency-weighted success</td>
</tr>
<tr>
<td align="left">Static Obstacle Collision</td>
<td align="left">Static obstacle collisions count</td>
</tr>
<tr>
<td align="left">Velocity Metrics</td>
<td align="left">Min, max, and avg. velocity</td>
</tr>
<tr>
<td align="left">Speed efficiency (<xref ref-type="bibr" rid="B94">Fraichard and Levesy, 2020</xref>)</td>
<td align="left">Ratio of nominal and actual speed</td>
</tr>
<tr>
<td align="left">Acceleration Metrics</td>
<td align="left">Min, max, and avg. acceleration</td>
</tr>
<tr>
<td align="left">Jerk Metrics</td>
<td align="left">Min, max, and avg. jerk</td>
</tr>
<tr>
<td align="left">Obstacle Distance</td>
<td align="left">Min, max, and avg. distance to obstacle</td>
</tr>
<tr>
<td align="left">Path Irregularity (<xref ref-type="bibr" rid="B111">Guzzi et al., 2013</xref>)</td>
<td align="left">Total deviation from a straight path</td>
</tr>
<tr>
<td align="left">Topological complexity (<xref ref-type="bibr" rid="B195">Mavrogiannis et al., 2018</xref>)</td>
<td align="left">Amount of entanglement within trajectory</td>
</tr>
<tr>
<td align="left">Path Efficiency</td>
<td align="left">Actual vs. straight-line ratio</td>
</tr>
<tr>
<td align="left">Failure To Progress (<xref ref-type="bibr" rid="B95">Francis et al., 2023</xref>)</td>
<td align="left">Goal progress failure over time period</td>
</tr>
<tr>
<td align="left">Human Collision</td>
<td align="left">Number of collisions with humans</td>
</tr>
<tr>
<td align="left">Social Distance</td>
<td align="left">Min, max, and avg. distance to humans</td>
</tr>
<tr>
<td align="left">Min Time To Collide (<xref ref-type="bibr" rid="B95">Francis et al., 2023</xref>)</td>
<td align="left">Min time to projected human collision</td>
</tr>
<tr>
<td align="left">Crowd Density</td>
<td align="left">Mean/max crowd density around robot</td>
</tr>
<tr>
<td align="left">Virtual Collision (<xref ref-type="bibr" rid="B227">Paez-Granados et al., 2022</xref>)</td>
<td align="left">Virtual boundary violations count</td>
</tr>
<tr>
<td align="left">Personal Space (<xref ref-type="bibr" rid="B318">Wang et al., 2022c</xref>)</td>
<td align="left">Avg time in minimum personal space</td>
</tr>
<tr>
<td align="left">Legibility (<xref ref-type="bibr" rid="B71">Dragan et al., 2013</xref>)</td>
<td align="left">Goal matches human expectation given robot motion</td>
</tr>
<tr>
<td align="left">Predictability (<xref ref-type="bibr" rid="B71">Dragan et al., 2013</xref>)</td>
<td align="left">Motion matches human expectation given known goal</td>
</tr>
<tr>
<td align="left">Projected Path (<xref ref-type="bibr" rid="B318">Wang et al., 2022c</xref>)</td>
<td align="left">Avg duration of path overlap with pedestrian</td>
</tr>
<tr>
<td align="left">Following Rate (<xref ref-type="bibr" rid="B241">Puig et al., 2023</xref>)</td>
<td align="left">Steps with maintained distance in human-follow task</td>
</tr>
<tr>
<td align="left">SPS (<xref ref-type="bibr" rid="B241">Puig et al., 2023</xref>)</td>
<td align="left">Path-weighted finding success in human find and follow task</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Social navigation benchmarking</title>
<p>This section benchmarks state-of-the-art social navigation planners from 7 categories, assessing their performance in realistic and challenging scenarios. We achieve efficient and consistent training and evaluation processes by leveraging GPU-based simulation. Additionally, planners are adapted to handle static obstacles such as walls, as most planners only process human positions. We benchmark each planner over 6 scenarios to provide insights into the strengths, limitations, and real-world applicability.</p>
<sec id="s4-1">
<label>4.1</label>
<title>Benchmark setup</title>
<p>A significant challenge in learning-based robotics, including social navigation, is the demanding computational cost of training and evaluation. To address this, we developed a benchmark that leverages GPU parallel computing to accelerate simulation and computation, significantly reducing training time and enabling more extensive experimentation and efficient benchmarking of social navigation planners.</p>
<p>The benchmark comprises three main components: kinematic motion simulation, sensor simulation, and crowd behavior modeling. Kinematic simulation is fully implemented on the GPU, including all computations for rewards and metrics, allowing efficient calculation of agent positions with respect to the map and robot frame. Sensor simulation is also performed on the GPU using Habitat Sim (<xref ref-type="bibr" rid="B270">Savva et al., 2019</xref>), which supports RGB and depth camera emulation (see <xref ref-type="fig" rid="F3">Figures 3e,f</xref>), and we generate 2D LiDAR observations via ray casting. The Habitat 3.0 (<xref ref-type="bibr" rid="B241">Puig et al., 2023</xref>) codebase further enables photorealistic rendering of 3D moving humans at high frame rates, achieving around 600 FPS for crowds of 40 humans. Existing crowd behavior models are primarily CPU-based, relying on well-established libraries. For diversity and robustness, we incorporate two models: SFM, using the implementation from <xref ref-type="bibr" rid="B97">Gao (2025)</xref> with parameters from <xref ref-type="bibr" rid="B121">Helbing et al. (2005)</xref>, and ORCA, using the implementation from <xref ref-type="bibr" rid="B285">St&#xfc;vel (2025)</xref> and parameters based on <xref ref-type="bibr" rid="B47">Chen C. et al. (2019)</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Top-down illustrations of the navigation scenarios <bold>(a&#x2013;d)</bold>, where the robot is shown in blue, the goal in green, and humans in red. Example RGB and depth images from the benchmark rendered using Habitat Sim are shown in <bold>(e,f)</bold>.</p>
</caption>
<graphic xlink:href="frobt-12-1658643-g003.tif">
<alt-text content-type="machine-generated">(a) Diagram showing an L-shaped path with two colored dots. (b) Diagram displaying a rectangular path with clusters of dots. (c) Straight path with scattered dots. (d) Cross-shaped path with clustered dots. (e) 3D rendering of people walking on an open sandy area. (f) Silhouettes of people walking in a foggy, dimly lit environment.</alt-text>
</graphic>
</fig>
<p>To enhance training efficiency and success, we employed curriculum learning during training. This technique gradually increases the difficulty of the scenarios as the robot improves. Initially, training focuses on less challenging configurations. As the training progresses, parameters such as crowd density and goal distance are systematically increased. Additionally, training focuses on scenarios where the robot performs poorly, ensuring that the robot performs well across all scenarios.</p>
<p>During evaluation, scenario parameters, including crowd density, goal distances, and map complexity, are randomly and uniformly sampled to ensure diverse testing conditions.</p>
</sec>
<sec id="s4-2">
<label>4.2</label>
<title>Benchmark scenarios</title>
<p>Benchmark scenarios, illustrated in <xref ref-type="fig" rid="F3">Figures 3a&#x2013;d</xref>, are designed to comprehensively evaluate social navigation by simulating a range of real-world challenges robots may encounter in crowded indoor and outdoor environments (<xref ref-type="bibr" rid="B98">Gao and Huang, 2022</xref>; <xref ref-type="bibr" rid="B95">Francis et al., 2023</xref>; <xref ref-type="bibr" rid="B283">Stratton et al., 2024</xref>). The robot&#x2019;s start and goal positions, environment size, and crowd density are randomized within defined bounds, with safety constraints to avoid infeasible or unsafe initializations. six representative scenarios are included: a static scenario with only obstacles to test navigation in narrow spaces; a doorway scenario evaluating interactions at chokepoints (<xref ref-type="bibr" rid="B277">Singamaneni et al., 2022</xref>); a corridor scenario capturing integration into unidirectional or bidirectional crowd flows; an intersection scenario representing complex areas where two flows meet; and open space scenarios that simulate unconstrained environments using both random and data-driven human motion, incorporating realistic pedestrian behavior from ETH (<xref ref-type="bibr" rid="B233">Pellegrini et al., 2009</xref>) and UCY (<xref ref-type="bibr" rid="B156">Lerner et al., 2007</xref>) datasets.</p>
</sec>
<sec id="s4-3">
<label>4.3</label>
<title>Benchmark planners</title>
<p>To evaluate social navigation strategies, we selected planners based on relevance, novelty, performance, and available implementations. The benchmark features three baselines and six learning-based planners, each covering a distinct social navigation category, along with an imitation-learning method. Since many learning-based planners do not natively handle static obstacles, we extend them with a LiDAR network (<xref ref-type="bibr" rid="B84">Fan et al., 2020</xref>), ensuring fair evaluation in environments with both dynamic and static obstacles.</p>
<sec id="s4-3-1">
<label>4.3.1</label>
<title>Baseline planners</title>
<p>The baseline planners include ORCA (<xref ref-type="bibr" rid="B303">Van den Berg et al., 2008</xref>), the SFM (<xref ref-type="bibr" rid="B120">Helbing and Molnar, 1995</xref>), and DWA (<xref ref-type="bibr" rid="B93">Fox et al., 1997</xref>). They serve as classical foundations for comparison with advanced methods. Each is given privileged access to the map layout and all human positions, ensuring optimal performance under ideal conditions.</p>
</sec>
<sec id="s4-3-2">
<label>4.3.2</label>
<title>End-to-end planner</title>
<p>The end-to-end planner is based on the RL policy from <xref ref-type="bibr" rid="B84">Fan et al. (2020)</xref>, which processes recent 2D LiDAR scans with a 1D CNN, combines them with the robot&#x2019;s state, and uses an MLP for action selection. Due to its suboptimal performance, we adopt an RNN-enhanced architecture (<xref ref-type="bibr" rid="B124">Hoeller et al., 2021</xref>), where the CNN output and robot state are fed into a GRU network, improving results. This end-to-end model learns navigation directly from sensor data, without using human state information.</p>
</sec>
<sec id="s4-3-3">
<label>4.3.3</label>
<title>Imitation learning-based planner</title>
<p>We implement Behavioral Cloning (BC) for imitation learning, offering a simple alternative to methods like GAIL (<xref ref-type="bibr" rid="B123">Ho and Ermon, 2016</xref>) without needing simulated environments. Trained on 35,000 successful human attention-based planner episodes, matching the planner&#x2019;s performance would signal robust generalization from real-world data. The network architecture mirrors the human attention-based planner.</p>
<sec id="s4-3-3-1">
<label>4.3.3.1</label>
<title>Human position-based planner</title>
<p>The GA3C-CADRL (<xref ref-type="bibr" rid="B80">Everett et al., 2018</xref>) planner uses an actor-critic policy with an LSTM to process human positions and velocities. We extend it with a LiDAR network (<xref ref-type="bibr" rid="B84">Fan et al., 2020</xref>) for static obstacle handling, enabling navigation in mixed environments. The LSTM input is zero-padded, and in scenarios without humans, the LSTM layer is skipped.</p>
</sec>
</sec>
<sec id="s4-3-4">
<label>4.3.4</label>
<title>Human attention-based planner</title>
<p>The SARL planner (<xref ref-type="bibr" rid="B47">Chen C. et al., 2019</xref>) employs an attention-based network to model robot-human attentions. We extend the original value network to an actor-critic framework and add a LiDAR network (<xref ref-type="bibr" rid="B84">Fan et al., 2020</xref>) for static obstacle handling. Unlike Liu et al. (<xref ref-type="bibr" rid="B168">Liu L. et al., 2020</xref>), which switches between separate policies for human and non-human scenarios, our approach uses a learned embedding to pad human input when no humans are present.</p>
</sec>
<sec id="s4-3-5">
<label>4.3.5</label>
<title>Human prediction-based planner</title>
<p>The prediction-based planner adapts the RGL model (<xref ref-type="bibr" rid="B49">Chen C. et al., 2020</xref>), integrating robot state and LiDAR input to predict human trajectories in the robot frame. These predicted trajectories are processed by an actor-critic policy, following the SARL planner (<xref ref-type="bibr" rid="B47">Chen C. et al., 2019</xref>), to handle fixed-size trajectories. When no humans are present, a learned embedding pads the input for consistency.</p>
</sec>
<sec id="s4-3-6">
<label>4.3.6</label>
<title>Safety-aware planner</title>
<p>Inspired by <xref ref-type="bibr" rid="B164">Linh et al. (2022)</xref>, the safety-aware planner combines ORCA (<xref ref-type="bibr" rid="B303">Van den Berg et al., 2008</xref>) for static environments and the human attention-based planner for dynamic settings, using a policy switcher based on obstacle proximity. This hybrid approach balances safety and efficiency by adapting to both static and human-dense scenarios.</p>
</sec>
</sec>
<sec id="s4-4">
<label>4.4</label>
<title>Results</title>
<p>Across six scenarios, learning-based planners consistently outperform model-based methods. In terms of success rate and safety, many of these learned policies consistently outperform traditional approaches. Unlike model-based planners, which prioritize obstacle avoidance, learning-based planners tend to emphasize maintaining a safe distance from humans, as illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Comparison success rate of of planners based on average minimum obstacle distance and minimum human distance.</p>
</caption>
<graphic xlink:href="frobt-12-1658643-g004.tif">
<alt-text content-type="machine-generated">Scatter plot showing various planning strategies represented as colored markers. The x-axis is obstacle distance (meters) from 0.35 to 0.75, and the y-axis is human distance (meters) from 0.2 to 0.375. A curve connects markers from the top right to bottom left. Each strategy has a unique color and marker shape as indicated in the legend on the left, which includes: SFM, ORCA, DWA, End-to-End, Imitation Learning, Human-Pose, Human-Interaction, Prediction Planner, and Safe Planner.</alt-text>
</graphic>
</fig>
<p>In the static scenario, all methods avoid collisions, so success rate and runtime distinguish performance. Model-based planners like ORCA achieve high success but are slower, while learning-based planners are overall faster, sometimes at the expense of a higher timeout rate. Imitation Learning struggles to generalize here. In the doorway scenario, where human-robot interactions are frequent, learning-based planners adapt better, leading to safer navigation and fewer collisions.</p>
<p>In the corridor scenario, both model-based and learning-based planners perform comparably, managing high success rates, efficiency, and safety distances. In contrast, in the intersection scenario, learning-based methods, particularly the prediction-based planner, achieve higher success rates.</p>
<p>In the open space random scenario, learning-based planners achieve higher success rates and smoother navigation by adapting to dynamic human movement, reducing congestion. Model-based methods, while faster, incur more collisions due to riskier behavior. This pattern holds across most scenarios as shown in <xref ref-type="fig" rid="F5">Figures 5a,c</xref>. In the open space data-driven scenario, learning-based planners remain safer while matching the running times of model-based approaches.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Average of planners based on success rate of each planner versus <bold>(a)</bold> running time, <bold>(b)</bold> robot velocity, and <bold>(c)</bold> path ratio.</p>
</caption>
<graphic xlink:href="frobt-12-1658643-g005.tif">
<alt-text content-type="machine-generated">Three graphs display evaluations of different methods. Graph (a) plots Running Time (seconds) against Success Rate (%), with varied markers and colors representing ten methods. Graph (b) shows Robot Velocity (meters/second) against Success Rate (%). Graph (c) presents Path Ratio versus Success Rate (%). Each graph includes a legend identifying methods like SFM and ORCA.</alt-text>
</graphic>
</fig>
<p>Among learning-based methods, end-to-end RL is notably conservative and prioritizes safety. Imitation Learning generalizes well in open spaces but struggles in constrained settings. The human position-based planner excels in open areas through direct spatial awareness, while the Human attention-based planner adapts best in crowded environments using attention mechanisms. The safety-aware planner balances efficiency and safety but remains limited by its learning-based component. The prediction-based planner, with its prediction module and expressive architecture, achieves the highest overall success rate and velocity, as shown in <xref ref-type="fig" rid="F5">Figure 5b</xref>.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Discussion and future directions</title>
<p>Despite the progress in social navigation, several challenges remain for learning-based social navigation to achieve safe and reliable real-world deployment. We organize this discussion around three priority levels: foundational requirements for safety and robustness, socially aligned behaviors for human acceptance, and capabilities that improve transparency and versatility.</p>
<sec id="s5-1">
<label>5.1</label>
<title>Foundations for safety and realism</title>
<p>Ensuring safe and robust navigation is the highest priority for real-world deployment.</p>
<sec id="s5-1-1">
<label>5.1.1</label>
<title>Safety and robustness</title>
<p>Most safety-oriented planners, like multi-policy approaches (<xref ref-type="bibr" rid="B286">Sun et al., 2019</xref>; <xref ref-type="bibr" rid="B140">Katyal et al., 2020</xref>; <xref ref-type="bibr" rid="B84">Fan et al., 2020</xref>), assume that reducing speed enhances safety, but this is not always valid; rapid maneuvers may be needed in dynamic, crowded settings. Relying solely on speed reduction can compromise safety in complex environments. Similarly, several human-prediction-based methods (<xref ref-type="bibr" rid="B342">Yao et al., 2024</xref>; <xref ref-type="bibr" rid="B361">Zhu et al., 2025</xref>) primarily forecast human motion without explicitly modeling the robot&#x2019;s influence on the crowd, which limits their ability to generate safe and adaptive plans. Instead, planners should learn context-aware safe behaviors, adjusting speed as needed and responding to emergencies, to achieve both safety and efficiency without unnecessary conservativeness.</p>
</sec>
<sec id="s5-1-2">
<label>5.1.2</label>
<title>Scenario diversity and generalization</title>
<p>A major limitation to model generalizability is the limited diversity of training scenarios. Future benchmarks should incorporate a wider range of realistic, data-driven scenarios reflecting true pedestrian distributions and start-goal configurations. Long-term crowd tracking datasets similar to, or even larger than ATC (<xref ref-type="bibr" rid="B28">Br&#x161;&#x10d;i&#x107; et al., 2013</xref>), which capture varied environments in a shopping mall, can help provide such diversity.</p>
</sec>
<sec id="s5-1-3">
<label>5.1.3</label>
<title>Physics and sensor realism</title>
<p>Physics simulators range from simple kinematic to detailed dynamic models, with high-fidelity simulation improving sim-to-real transfer and enabling robot-specific planners that can integrate low-level control, such as direct wheel velocities. Likewise, accurate sensor simulation enhances robustness; while most simulators use generic models for simplicity (<xref ref-type="bibr" rid="B130">Inc D, 2025</xref>), sensor-specific models that replicate real-world parameters and noise can significantly improve generalization and sim-to-real performance.</p>
</sec>
<sec id="s5-1-4">
<label>5.1.4</label>
<title>Realistic crowd simulation</title>
<p>Most crowd simulation methods focus on human-human and human-obstacle interactions, but accurately modeling human-robot interactions remains a challenge. Some approaches ignore the robot&#x2019;s presence (<xref ref-type="bibr" rid="B48">Chen Y. et al., 2020</xref>; <xref ref-type="bibr" rid="B73">Dugas et al., 2021</xref>), leading to unrealistic and overly conservative behavior, while others treat robots as humans or add randomness for robustness (<xref ref-type="bibr" rid="B47">Chen C. et al., 2019</xref>; <xref ref-type="bibr" rid="B283">Stratton et al., 2024</xref>). However, these do not fully capture the diverse ways humans respond to robots, which depend on robot-specific factors like size, shape, and movement. More advanced crowd models that reflect these characteristics are needed for realistic social navigation simulation.</p>
</sec>
<sec id="s5-1-5">
<label>5.1.5</label>
<title>Robust evaluation</title>
<p>Advancing social navigation requires robust benchmarking methods that can accurately represent the planner&#x2019;s performance. Key directions include adopting realistic crowd simulation, conducting real-world evaluations, and refining social metrics (<xref ref-type="bibr" rid="B95">Francis et al., 2023</xref>; <xref ref-type="bibr" rid="B98">Gao and Huang, 2022</xref>). Automated, objective real-world evaluation frameworks are increasingly important, as subjective user feedback is impractical to standardize. Future evaluations could use objective, non-verbal indicators, such as body language or facial expressions to better assess human comfort and social acceptance, ensuring planners are both effective and socially appropriate.</p>
</sec>
</sec>
<sec id="s5-2">
<label>5.2</label>
<title>Social alignment and preferences</title>
<p>Beyond safety, social navigation must align with human expectations and adapt to cultural and individual differences.</p>
<sec id="s5-2-1">
<label>5.2.1</label>
<title>Social norms and compliance</title>
<p>Social norms are informal rules guiding behavior in shared spaces, extending beyond collision avoidance and proxemics (<xref ref-type="bibr" rid="B115">Hall, 1963</xref>). For instance, smoothly avoiding social groups is addressed by some crowd prediction methods (<xref ref-type="bibr" rid="B21">Bisagno et al., 2018</xref>; <xref ref-type="bibr" rid="B89">Fernando et al., 2019</xref>), but is incorporated into only a few navigation algorithms (<xref ref-type="bibr" rid="B20">Bhaskara et al., 2023</xref>). Other norms, such as culturally specific conventions (<xref ref-type="bibr" rid="B43">Chen et al., 2017c</xref>), are context-sensitive and not universal, suggesting the value of learning social norms directly from large-scale crowd data rather than relying solely on handcrafted heuristics. Vision-language models (VLMs) open an additional pathway by enabling robots to ground these norms in natural language, reason about complex social contexts, and even communicate intentions to humans in interpretable ways. Effective social navigation will likely require a combination of data-driven norm learning and VLM-based reasoning, alongside intention communication that may be verbal (<xref ref-type="bibr" rid="B72">Dugas et al., 2020</xref>; <xref ref-type="bibr" rid="B220">Nishimura and Yonetani, 2020</xref>) or conveyed through non-verbal cues, as highlighted in autonomous vehicle research (<xref ref-type="bibr" rid="B113">Habibovic et al., 2018</xref>).</p>
</sec>
<sec id="s5-2-2">
<label>5.2.2</label>
<title>Human preferences</title>
<p>Social navigation is not a one-size-fits-all solution. Individuals and crowds vary in preferred comfort distance, speed, and interaction style. Future work should emphasize preference-aware navigation, where robots learn and adapt to individual users or cultural groups, potentially combining reinforcement learning with preference learning, feedback, or large language models that capture human expectations and feedback. Although current approaches consider human preferences during training (<xref ref-type="bibr" rid="B56">Choi et al., 2020</xref>), accommodating post-deployment feedback and achieving continuous learning remain open challenges.</p>
</sec>
</sec>
<sec id="s5-3">
<label>5.3</label>
<title>Transparency and reasoning</title>
<p>To ensure long-term acceptance, learning-based systems must be interpretable, communicative, and capable of reasoning based on context.</p>
<sec id="s5-3-1">
<label>5.3.1</label>
<title>Explainability and transparency</title>
<p>A major challenge in learning-based planners is the difficulty of interpreting the reasoning behind their decisions, which is often referred to as <italic>explainability</italic> (<xref ref-type="bibr" rid="B309">Vouros, 2022</xref>). Integrating explainability improves user trust, allows better debugging, and clarifies the decision-making process. Several techniques exist, such as <italic>saliency maps</italic>, which visually indicate influential regions within image-based inputs (<xref ref-type="bibr" rid="B129">Huber et al., 2021</xref>), and approaches that provide verbal explanations for their decisions (<xref ref-type="bibr" rid="B72">Dugas et al., 2020</xref>). Integrating these explainability methods into learning-based social navigation can create more transparent, interpretable, and user-friendly systems.</p>
</sec>
<sec id="s5-3-2">
<label>5.3.2</label>
<title>Social vision-language navigation</title>
<p>Recent advances in vision-language navigation (VLNs) (<xref ref-type="bibr" rid="B5">An et al., 2022</xref>) highlight opportunities to enrich social navigation with multimodal reasoning capabilities and improve functional versatility. Beyond instruction following (<xref ref-type="bibr" rid="B6">Anderson et al., 2018a</xref>), VLNs can support a wide range of tasks such as visual question answering (<xref ref-type="bibr" rid="B326">Wu et al., 2024</xref>), describing social situations, or embodied dialog (<xref ref-type="bibr" rid="B114">Hahn et al., 2020</xref>). Social VLN could allow robots to interpret human intent, infer social norms from linguistic context, and communicate their own decisions in interpretable ways.</p>
</sec>
</sec>
</sec>
</body>
<back>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>RA: Writing &#x2013; original draft, Software, Writing &#x2013; review and editing. CC: Writing &#x2013; review and editing. RR: Writing &#x2013; review and editing. DP-G: Writing &#x2013; review and editing, Methodology.</p>
</sec>
<ack>
<title>Acknowledgements</title>
<p>We acknowledge the support of S. Dey for providing feedback on the initial draft of the manuscript. ChatGPT-4 was used to assist with grammar checks and basic fact-checking in this review.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. ChatGPT-4 was used to assist with grammar checks and basic fact-checking in this review.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frobt.2025.1658643/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frobt.2025.1658643/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="n1" fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2895537/overview">Allan Wang</ext-link>, Miraikan &#x2013; The National Museum of Emerging Science and Innovation, Japan</p>
</fn>
<fn id="n2" fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/666080/overview">Suresh Kumaar Jayaraman</ext-link>, Carnegie Mellon University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3140660/overview">Yigit Yildirim</ext-link>, Bogazici Universitesi Muhendislik Fakultesi, T&#xfc;rkiye</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Achiam</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Held</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tamar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Constrained policy optimization</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>. <publisher-loc>Sydney, Australia</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>22</fpage>&#x2013;<lpage>31</lpage>.</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Alahi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goel</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ramanathan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Robicquet</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fei-Fei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Savarese</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Social lstm: human trajectory prediction in crowded spaces</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>961</fpage>&#x2013;<lpage>971</lpage>.</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Alonso-Mora</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Breitenmoser</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rufli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Beardsley</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Optimal reciprocal collision avoidance for multiple non-holonomic robots</article-title>,&#x201d; in <source>Distributed autonomous robotic systems: the 10th international symposium</source>. <publisher-name>Springer</publisher-name>, <fpage>203</fpage>&#x2013;<lpage>216</lpage>.</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Amano</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kato</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Autonomous Mobile robot navigation for complicated environments by switching multiple control policies</article-title>,&#x201d; in <source>
<italic>IECON 2022&#x2013;48th annual conference of the IEEE industrial electronics Society</italic> (IEEE)</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>An</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Bevbert: multimodal map pre-training for language-guided navigation</article-title>. <source>arXiv Prepr. arXiv:2212.04385</source>.</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Anderson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Teney</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bruce</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>S&#xfc;nderhauf</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2018a</year>). &#x201c;<article-title>Vision-and-language navigation: interpreting visually-grounded navigation instructions in real environments</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>3674</fpage>&#x2013;<lpage>3683</lpage>.</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anderson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chaplot</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Dosovitskiy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Koltun</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2018b</year>). <article-title>On evaluation of embodied navigation agents</article-title>. <comment>
<italic>arXiv preprint arXiv:1807.06757</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Anderson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Shrivastava</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Truong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Majumdar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Parikh</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Batra</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Sim-to-real transfer for vision-and-language navigation</article-title>,&#x201d; in <source>
<italic>Conference on robot learning</italic> (PMLR)</source>, <fpage>671</fpage>&#x2013;<lpage>681</lpage>.</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Arjovsky</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chintala</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bottou</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Wasserstein generative adversarial networks</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>. <publisher-name>Sydney, Australia</publisher-name>: <publisher-name>PMLR</publisher-name>, <fpage>214</fpage>&#x2013;<lpage>223</lpage>.</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Aroor</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Esptein</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Korpan</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Mengeros: a crowd simulation tool for autonomous robot navigation</article-title>,&#x201d; in <source>AAAI fall symposium series</source>.</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Arras</surname>
<given-names>K. O.</given-names>
</name>
<name>
<surname>Mozos</surname>
<given-names>O. M.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2007</year>). &#x201c;<article-title>Using boosted features for the detection of people in 2d range data</article-title>,&#x201d; in <source>
<italic>Proceedings 2007 IEEE international conference on robotics and automation</italic> (IEEE)</source>, <fpage>3402</fpage>&#x2013;<lpage>3407</lpage>.</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bae</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Sit dataset: socially interactive pedestrian trajectory dataset for social navigation robots</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>.</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Bansal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bajcsy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ratner</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Dragan</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Tomlin</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A hamilton-jacobi reachability-based framework for predicting and analyzing human motion for safe planning</article-title>,&#x201d; in <source>
<italic>2020 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>7149</fpage>&#x2013;<lpage>7155</lpage>.</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bastani</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Pu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Solar-Lezama</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Verifiable reinforcement learning <italic>via</italic> policy extraction</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>31</volume>.</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Benfold</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Reid</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Stable multi-target tracking in real-time surveillance video</article-title>. <source>
<italic>CVPR 2011</italic> (IEEE)</source>, <fpage>3457</fpage>&#x2013;<lpage>3464</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr.2011.5995667</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Louradour</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Collobert</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Weston</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Curriculum learning</article-title>,&#x201d; in <source>Proceedings of the 26th annual international conference on machine learning</source>, <fpage>41</fpage>&#x2013;<lpage>48</lpage>.</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Bertoni</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kreiss</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mordan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Alahi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Monstereo: when monocular and stereo meet at the tail of 3d human localization</source>. <publisher-name>IEEE International Conference on Robotics and Automation ICRA</publisher-name>, <fpage>5126</fpage>&#x2013;<lpage>5132</lpage>.</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beyer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hermans</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Leibe</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Drow: Real-Time deep learning-based wheelchair detection in 2-d range data</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>2</volume>, <fpage>585</fpage>&#x2013;<lpage>592</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2016.2645131</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beyer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hermans</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Linder</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Arras</surname>
<given-names>K. O.</given-names>
</name>
<name>
<surname>Leibe</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep person detection in 2d range data</article-title>. <source>arXiv Prepr. arXiv:1804.02463</source>.</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Bhaskara</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chiu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bera</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Sg-lstm: social group lstm for robot navigation through dense crowds</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>3835</fpage>&#x2013;<lpage>3840</lpage>.</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Bisagno</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Conci</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Group lstm: group trajectory prediction in crowded scenarios</article-title>,&#x201d; in <source>Proceedings of the European conference on computer vision (ECCV) workshops</source>.</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Blundell</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cornebise</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kavukcuoglu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wierstra</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Weight uncertainty in neural network</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>. <publisher-loc>Lille, France</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>1613</fpage>&#x2013;<lpage>1622</lpage>.</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Bock</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Krajewski</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Moers</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Runde</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vater</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Eckstein</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>The ind dataset: a drone dataset of naturalistic road user trajectories at german intersections</article-title>,&#x201d; in <source>2020 IEEE intelligent vehicles symposium (IV)</source>. <publisher-name>IEEE</publisher-name>, <fpage>1929</fpage>&#x2013;<lpage>1934</lpage>.</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bojarski</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>End to end learning for self-driving cars</article-title>. <source>arXiv Prepr. arXiv:1604.07316</source>.</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brito</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Everett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Alonso-Mora</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Where to go next: learning a subgoal recommendation policy for navigation in dynamic environments</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>6</volume>, <fpage>4616</fpage>&#x2013;<lpage>4623</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2021.3068662</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Goo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Nagarajan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Niekum</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Extrapolating beyond suboptimal demonstrations <italic>via</italic> inverse reinforcement learning from observations</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>. <publisher-loc>Long Beach, California, United States</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>783</fpage>&#x2013;<lpage>792</lpage>.</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Goo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Niekum</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Better-than-demonstrator imitation learning <italic>via</italic> automatically-ranked demonstrations</article-title>,&#x201d; in <source>
<italic>Conference on robot learning</italic> (PMLR)</source>, <fpage>330</fpage>&#x2013;<lpage>359</lpage>.</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Br&#x161;&#x10d;i&#x107;</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kanda</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ikeda</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Miyashita</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Person tracking in large public spaces using 3-d range sensors</article-title>. <source>IEEE Trans. Human-Machine Syst.</source> <volume>43</volume>, <fpage>522</fpage>&#x2013;<lpage>534</lpage>. <pub-id pub-id-type="doi">10.1109/thms.2013.2283945</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Cremers</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Fox</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>H&#xe4;hnel</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lakemeyer</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Schulz</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>1999</year>). <article-title>Experiences with an interactive museum tour-guide robot</article-title>. <source>Artif. Intell.</source> <volume>114</volume>, <fpage>3</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1016/s0004-3702(99)00070-3</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Caesar</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bankiti</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Lang</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Vora</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liong</surname>
<given-names>V. E.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Nuscenes: a multimodal dataset for autonomous driving</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>11621</fpage>&#x2013;<lpage>11631</lpage>.</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campbell</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kulis</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Carin</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Dynamic clustering <italic>via</italic> asymptotics of the dependent dirichlet process mixture</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>26</volume>.</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Cancelli</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Campari</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Serafini</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>A. X.</given-names>
</name>
<name>
<surname>Ballan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Exploiting proximity-aware tasks for embodied social navigation</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>10957</fpage>&#x2013;<lpage>10967</lpage>.</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Carion</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Massa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Synnaeve</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Usunier</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kirillov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zagoruyko</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>End-to-end object detection with transformers</article-title>,&#x201d; in <source>European conference on computer vision</source>. <publisher-name>Springer</publisher-name>, <fpage>213</fpage>&#x2013;<lpage>229</lpage>.</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carlevaris-Bianco</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ushani</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Eustice</surname>
<given-names>R. M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>University of michigan north campus long-term vision and lidar dataset</article-title>. <source>Int. J. Robotics Res.</source> <volume>35</volume>, <fpage>1023</fpage>&#x2013;<lpage>1035</lpage>. <pub-id pub-id-type="doi">10.1177/0278364915614638</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Cmdfusion: bidirectional fusion network with cross-modality knowledge distillation for lidar semantic segmentation</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume>, <fpage>771</fpage>&#x2013;<lpage>778</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2023.3335771</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chandra</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bhattacharya</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Roncal</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Robusttp: end-to-end trajectory prediction for heterogeneous road-agents in dense traffic with noisy sensor inputs</article-title>,&#x201d; in <source>Proceedings of the 3rd ACM computer science in cars symposium</source>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>.</mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Funkhouser</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Halber</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Niessner</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Savva</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Matterport3d: learning from rgb-d data in indoor environments</article-title>. <source>arXiv Prepr. arXiv:1709.06158</source>, <fpage>667</fpage>&#x2013;<lpage>676</lpage>. <pub-id pub-id-type="doi">10.1109/3dv.2017.00081</pub-id>
</mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Charalampous</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kostavelis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Gasteratos</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Robot navigation in large-scale social maps: an action recognition approach</article-title>. <source>Expert Syst. Appl.</source> <volume>66</volume>, <fpage>261</fpage>&#x2013;<lpage>273</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2016.09.026</pub-id>
</mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Charalampous</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kostavelis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Gasteratos</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Recent trends in social aware robot navigation: a survey</article-title>. <source>Robotics Aut. Syst.</source> <volume>93</volume>, <fpage>85</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1016/j.robot.2017.03.002</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chavdarova</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Baqu&#xe9;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bouquet</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Maksai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jose</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bagautdinov</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Wildtrack: a multi-camera hd dataset for dense unscripted pedestrian detection</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>5030</fpage>&#x2013;<lpage>5039</lpage>.</mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Shuai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2017a</year>). <article-title>Robots serve humans in public places&#x2014;kejia robot as a shopping assistant</article-title>. <source>Int. J. Adv. Robotic Syst.</source> <volume>14</volume>, <fpage>172988141770356</fpage>. <pub-id pub-id-type="doi">10.1177/1729881417703569</pub-id>
</mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Everett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2017b</year>). &#x201c;<article-title>Decentralized non-communicating multiagent collision avoidance with deep reinforcement learning</article-title>,&#x201d; in <source>
<italic>2017 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>285</fpage>&#x2013;<lpage>292</lpage>.</mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>Everett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2017c</year>). <source>Socially aware motion planning with deep reinforcement learning</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>1343</fpage>&#x2013;<lpage>1350</lpage>.</mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2017d</year>). &#x201c;<article-title>Multi-view 3d object detection network for autonomous driving</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>1907</fpage>&#x2013;<lpage>1915</lpage>.</mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Robot navigation based on human trajectory prediction and multiple travel modes</article-title>. <source>Appl. Sci.</source> <volume>8</volume>, <fpage>2205</fpage>. <pub-id pub-id-type="doi">10.3390/app8112205</pub-id>
</mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>Mapless collaborative navigation for a multi-robot system based on the deep reinforcement learning</article-title>. <source>Appl. Sci.</source> <volume>9</volume>, <fpage>4198</fpage>. <pub-id pub-id-type="doi">10.3390/app9204198</pub-id>
</mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kreiss</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alahi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019b</year>). &#x201c;<article-title>Crowd-robot interaction: Crowd-aware robot navigation with attention-based deep reinforcement learning</article-title>,&#x201d; in <source>
<italic>2019 international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>6015</fpage>&#x2013;<lpage>6022</lpage>.</mixed-citation>
</ref>
<ref id="B48">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>B. E.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Robot navigation in crowds by graph convolutional networks with attention learned from human gaze</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>2754</fpage>&#x2013;<lpage>2761</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2020.2972868</pub-id>
</mixed-citation>
</ref>
<ref id="B49">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nikdel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mori</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Savva</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020b</year>). &#x201c;<article-title>Relational graph learning for crowd navigation</article-title>,&#x201d; in <source>2020 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source>. <publisher-name>IEEE</publisher-name>, <fpage>10007</fpage>&#x2013;<lpage>10013</lpage>.</mixed-citation>
</ref>
<ref id="B50">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Paleja</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gombolay</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Learning from suboptimal demonstration <italic>via</italic> self-supervised reward regression</article-title>,&#x201d; in <source>
<italic>Conference on robot learning</italic> (PMLR)</source>, <fpage>1262</fpage>&#x2013;<lpage>1277</lpage>.</mixed-citation>
</ref>
<ref id="B51">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>End-to-end 3d dense captioning with vote2cap-detr</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>11124</fpage>&#x2013;<lpage>11133</lpage>.</mixed-citation>
</ref>
<ref id="B52">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Multi-objective deep reinforcement learning for crowd-aware robot navigation with dynamic human preference</article-title>. <source>Neural Comput. Appl.</source> <volume>35</volume>, <fpage>16247</fpage>&#x2013;<lpage>16265</lpage>. <pub-id pub-id-type="doi">10.1007/s00521-023-08385-4</pub-id>
</mixed-citation>
</ref>
<ref id="B53">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chiang</surname>
<given-names>H. T. L.</given-names>
</name>
<name>
<surname>Faust</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fiser</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Francis</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Learning navigation behaviors end-to-end with autorl</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>4</volume>, <fpage>2007</fpage>&#x2013;<lpage>2014</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2019.2899918</pub-id>
</mixed-citation>
</ref>
<ref id="B54">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Shahid</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Savarese</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>What are they doing? collective activity classification using spatio-temporal relationship among people</article-title>,&#x201d; in <source>
<italic>2009 IEEE 12th international conference on computer vision workshops, ICCV workshops</italic> (IEEE)</source>, <fpage>1282</fpage>&#x2013;<lpage>1289</lpage>.</mixed-citation>
</ref>
<ref id="B55">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Seok</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep reinforcement learning of navigation in a complex and crowded environment with a limited field of view</article-title>. <source>
<italic>Int. Conf. Robotics Automation (ICRA)</italic> (IEEE)</source>, <fpage>5993</fpage>&#x2013;<lpage>6000</lpage>. <pub-id pub-id-type="doi">10.1109/icra.2019.8793979</pub-id>
</mixed-citation>
</ref>
<ref id="B56">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dance</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Je</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>Ks</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Seo</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Fast adaptation of deep reinforcement learning-based navigation skills to human preference</article-title>,&#x201d; in <source>
<italic>2020 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>3363</fpage>&#x2013;<lpage>3370</lpage>.</mixed-citation>
</ref>
<ref id="B57">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Christiano</surname>
<given-names>P. F.</given-names>
</name>
<name>
<surname>Leike</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Martic</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Legg</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Amodei</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Deep reinforcement learning from human preferences</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>30</volume>.</mixed-citation>
</ref>
<ref id="B58">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chuang</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>N. C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Hung</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y. W.</given-names>
</name>
<name>
<surname>Teng</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Deep trail-following robotic guide dog in pedestrian environments for people who are blind and visually impaired-learning from virtual and real worlds</article-title>,&#x201d; in <source>
<italic>2018 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>5849</fpage>&#x2013;<lpage>5855</lpage>.</mixed-citation>
</ref>
<ref id="B59">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Cong</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qiao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Stcrowd: a multimodal dataset for pedestrian perception in crowded scenes</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>19608</fpage>&#x2013;<lpage>19617</lpage>.</mixed-citation>
</ref>
<ref id="B60">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Costa</surname>
<given-names>E. D. S.</given-names>
</name>
<name>
<surname>Gouvea</surname>
<given-names>Jr M. M.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Autonomous navigation in dynamic environments with reinforcement learning and heuristic</article-title>,&#x201d; in <source>2010 ninth international conference on machine learning and applications</source>. <publisher-name>IEEE</publisher-name>, <fpage>37</fpage>&#x2013;<lpage>42</lpage>.</mixed-citation>
</ref>
<ref id="B61">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Cui</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Learning world transition model for socially aware robot navigation</article-title>,&#x201d; in <source>
<italic>2021 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>9262</fpage>&#x2013;<lpage>9268</lpage>.</mixed-citation>
</ref>
<ref id="B62">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Curtis</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Best</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Menge</surname>
<given-names>M. D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A modular framework for simulating crowd movement</article-title>. <source>Collect. Dyn.</source> <volume>1</volume>, <fpage>1</fpage>&#x2013;<lpage>40</lpage>.</mixed-citation>
</ref>
<ref id="B63">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Czarnecki</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jayakumar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jaderberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hasenclever</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Teh</surname>
<given-names>Y. W.</given-names>
</name>
<name>
<surname>Heess</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Mix and match agent curricula for reinforcement learning</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>. <publisher-loc>Stockholmsm&#xe4;ssan, Sweden</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>1087</fpage>&#x2013;<lpage>1095</lpage>.</mixed-citation>
</ref>
<ref id="B64">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Czarnecki</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Pascanu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Osindero</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jayakumar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Swirszcz</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Jaderberg</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Distilling policy distillation</article-title>,&#x201d; in <source>The 22nd International Conference on Artificial Intelligence and Statistics</source>. <publisher-loc>Okinawa, Japan</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>1331</fpage>&#x2013;<lpage>1340</lpage>.</mixed-citation>
</ref>
<ref id="B65">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dalal</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Triggs</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Histograms of oriented gradients for human detection</article-title>. <source>
<italic>2005 IEEE Comput. Soc. Conf. Comput. Vis. pattern Recognit. (CVPR&#x2019;05)</italic> (Ieee)</source> <volume>1</volume>, <fpage>886</fpage>&#x2013;<lpage>893</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr.2005.177</pub-id>
</mixed-citation>
</ref>
<ref id="B66">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Datseris</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Vahdati</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>DuBois</surname>
<given-names>T. C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Agents. jl: a performant and feature-full agent-based modeling software of minimal code complexity</article-title>. <source>Simulation</source> <volume>100</volume>, <fpage>1019</fpage>&#x2013;<lpage>1031</lpage>. <pub-id pub-id-type="doi">10.1177/00375497211068820</pub-id>
</mixed-citation>
</ref>
<ref id="B67">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>de Heuvel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Corral</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bruckschen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bennewitz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Learning personalized human-aware robot navigation using virtual reality demonstrations from a user study</article-title>,&#x201d; in <source>2022 31st IEEE international conference on robot and human interactive communication (RO-MAN)</source>. <publisher-name>IEEE</publisher-name>, <fpage>898</fpage>&#x2013;<lpage>905</lpage>.</mixed-citation>
</ref>
<ref id="B68">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>de Heuvel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Corral</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kreis</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Conradi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Driemel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bennewitz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Learning depth vision-based personalized robot navigation from dynamic demonstrations in virtual reality</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>6757</fpage>&#x2013;<lpage>6764</lpage>.</mixed-citation>
</ref>
<ref id="B69">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Heuvel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sethuraman</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bennewitz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Learning adaptive multi-objective robot navigation with demonstrations</article-title>. <comment>
<italic>arXiv preprint arXiv:2404.04857</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B70">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dequaire</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ondr&#xfa;&#x161;ka</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Posner</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep tracking in the wild: end-to-end tracking using recurrent neural networks</article-title>. <source>Int. J. Robotics Res.</source> <volume>37</volume>, <fpage>492</fpage>&#x2013;<lpage>512</lpage>. <pub-id pub-id-type="doi">10.1177/0278364917710543</pub-id>
</mixed-citation>
</ref>
<ref id="B71">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dragan</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Srinivasa</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Legibility and predictability of robot motion</article-title>. <source>
<italic>8th ACM/IEEE Int. Conf. Human-Robot Interact. (HRI)</italic> (IEEE)</source>, <fpage>301</fpage>&#x2013;<lpage>308</lpage>. <pub-id pub-id-type="doi">10.1109/hri.2013.6483603</pub-id>
</mixed-citation>
</ref>
<ref id="B72">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Dugas</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nieto</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Ian: multi-behavior navigation planning for robots in real, crowded environments</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>11368</fpage>&#x2013;<lpage>11375</lpage>.</mixed-citation>
</ref>
<ref id="B73">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Dugas</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nieto</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Navrep: unsupervised representations for reinforcement learning of robot navigation in dynamic human environments</article-title>,&#x201d; in <source>
<italic>2021 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>7829</fpage>&#x2013;<lpage>7835</lpage>.</mixed-citation>
</ref>
<ref id="B74">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dynamics</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2025</year>). <source>Jupedsim</source>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/PedestrianDynamics/jupedsim">https://github.com/PedestrianDynamics/jupedsim</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B75">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Echeverria</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lassabe</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Degroote</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lemaignan</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Modular open robots simulation engine: morse</article-title>,&#x201d; in <source>
<italic>2011 ieee international conference on robotics and automation</italic> (IEEE)</source>, <fpage>46</fpage>&#x2013;<lpage>51</lpage>.</mixed-citation>
</ref>
<ref id="B76">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Eiffert</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pirmarzdashti</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sukkarieh</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020a</year>). &#x201c;<article-title>Path planning in dynamic environments using generative rnns and monte carlo tree search</article-title>,&#x201d; in <source>
<italic>2020 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>10263</fpage>&#x2013;<lpage>10269</lpage>.</mixed-citation>
</ref>
<ref id="B77">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eiffert</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Worrall</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sukkarieh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nebot</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Probabilistic crowd gan: multimodal pedestrian trajectory prediction using a graph vehicle-pedestrian attention network</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>5026</fpage>&#x2013;<lpage>5033</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2020.3004324</pub-id>
</mixed-citation>
</ref>
<ref id="B78">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Eppenberger</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cesari</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dymczyk</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Dub&#xe9;</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Leveraging stereo-camera data for real-time dynamic obstacle detection and tracking</article-title>,&#x201d; in <source>
<italic>IEEE/RSJ international conference on intelligent robots and systems (IROS)</italic> (IEEE)</source>, <fpage>10528</fpage>&#x2013;<lpage>10535</lpage>.</mixed-citation>
</ref>
<ref id="B79">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Ettinger</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Caine</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pradhan</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Large scale interactive motion forecasting for autonomous driving: the waymo open motion dataset</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>9710</fpage>&#x2013;<lpage>9719</lpage>.</mixed-citation>
</ref>
<ref id="B80">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Everett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Motion planning among dynamic, decision-making agents with deep reinforcement learning</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>3052</fpage>&#x2013;<lpage>3059</lpage>.</mixed-citation>
</ref>
<ref id="B81">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Everett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Collision avoidance in pedestrian-rich environments with deep reinforcement learning</article-title>. <source>Ieee Access</source> <volume>9</volume>, <fpage>10357</fpage>&#x2013;<lpage>10377</lpage>. <pub-id pub-id-type="doi">10.1109/access.2021.3050338</pub-id>
</mixed-citation>
</ref>
<ref id="B82">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Fahad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Learning how pedestrians navigate: a deep inverse reinforcement learning approach</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>819</fpage>&#x2013;<lpage>826</lpage>.</mixed-citation>
</ref>
<ref id="B83">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Crowdmove: autonomous mapless navigation in crowded scenarios</article-title>. <comment>
<italic>arXiv preprint arXiv:1807.07870</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B84">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Distributed multi-robot collision avoidance <italic>via</italic> deep reinforcement learning for navigation in complex scenarios</article-title>. <source>Int. J. Robotics Res.</source> <volume>39</volume>, <fpage>856</fpage>&#x2013;<lpage>892</lpage>. <pub-id pub-id-type="doi">10.1177/0278364920916531</pub-id>
</mixed-citation>
</ref>
<ref id="B85">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Faure</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Cromosim</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cromosim.fr">https://www.cromosim.fr</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B86">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Faust</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Oslund</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ramirez</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Francis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tapia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fiser</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Prm-rl: long-range robotic navigation tasks by combining reinforcement learning and sampling-based planning</article-title>,&#x201d; in <source>2018 IEEE international conference on robotics and automation (ICRA)</source>. <publisher-name>IEEE</publisher-name>, <fpage>5113</fpage>&#x2013;<lpage>5120</lpage>.</mixed-citation>
</ref>
<ref id="B87">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Felzenszwalb</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>McAllester</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ramanan</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2008</year>). &#x201c;<article-title>A discriminatively trained, multiscale, deformable part model</article-title>,&#x201d; in <source>
<italic>2008 IEEE conference on computer vision and pattern recognition</italic> (Ieee)</source>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</mixed-citation>
</ref>
<ref id="B88">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fernando</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Denman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sridharan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fookes</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Soft&#x2b; hardwired attention: an lstm framework for human trajectory prediction and abnormal event detection</article-title>. <source>Neural Netw.</source> <volume>108</volume>, <fpage>466</fpage>&#x2013;<lpage>478</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2018.09.002</pub-id>
<pub-id pub-id-type="pmid">30317132</pub-id>
</mixed-citation>
</ref>
<ref id="B89">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Fernando</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Denman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sridharan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fookes</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Gd-gan: generative adversarial networks for trajectory prediction and group detection in crowds</article-title>,&#x201d; in <source>Computer Vision&#x2013;ACCV 2018: 14th Asian conference on computer vision, Perth, Australia, December 2&#x2013;6, 2018, revised selected papers, part I 14</source>. <publisher-name>Springer</publisher-name>, <fpage>314</fpage>&#x2013;<lpage>330</lpage>.</mixed-citation>
</ref>
<ref id="B90">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferrer</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zulueta</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Cotarelo</surname>
<given-names>F. H.</given-names>
</name>
<name>
<surname>Sanfeliu</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Robot social-aware navigation framework to accompany people walking side-by-side</article-title>. <source>Aut. robots</source> <volume>41</volume>, <fpage>775</fpage>&#x2013;<lpage>793</lpage>. <pub-id pub-id-type="doi">10.1007/s10514-016-9584-y</pub-id>
</mixed-citation>
</ref>
<ref id="B91">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Deep visual foresight for planning robot motion</article-title>,&#x201d; in <source>
<italic>2017 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>2786</fpage>&#x2013;<lpage>2793</lpage>.</mixed-citation>
</ref>
<ref id="B92">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Florensa</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Held</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Geng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Automatic goal generation for reinforcement learning agents</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>. <publisher-loc>Stockholmsm&#xe4;ssan, Sweden</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>1515</fpage>&#x2013;<lpage>1528</lpage>.</mixed-citation>
</ref>
<ref id="B93">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fox</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Thrun</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>The dynamic window approach to collision avoidance</article-title>. <source>IEEE Robotics and Automation Mag.</source> <volume>4</volume>, <fpage>23</fpage>&#x2013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.1109/100.580977</pub-id>
</mixed-citation>
</ref>
<ref id="B94">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fraichard</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Levesy</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>From crowd simulation to robot navigation in crowds</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>729</fpage>&#x2013;<lpage>735</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2020.2965032</pub-id>
</mixed-citation>
</ref>
<ref id="B95">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Francis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>P&#xe9;rez-d&#x2019;Arpino</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Alahi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alami</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Principles and guidelines for evaluating social robot navigation algorithms</article-title>. <source>arXiv Prepr. arXiv:2306.16740</source>.</mixed-citation>
</ref>
<ref id="B96">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Gal</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ghahramani</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Dropout as a bayesian approximation: representing model uncertainty in deep learning</article-title>,&#x201d; in <source>
<italic>International conference on machine learning</italic> (PMLR)</source>, <fpage>1050</fpage>&#x2013;<lpage>1059</lpage>.</mixed-citation>
</ref>
<ref id="B97">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Pysocialforce</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/yuxiang-gao/PySocialForce">https://github.com/yuxiang-gao/PySocialForce</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B98">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>C. M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Evaluation of socially-aware robot navigation</article-title>. <source>Front. Robotics AI</source> <volume>8</volume>, <fpage>721317</fpage>. <pub-id pub-id-type="doi">10.3389/frobt.2021.721317</pub-id>
<pub-id pub-id-type="pmid">35096978</pub-id>
</mixed-citation>
</ref>
<ref id="B99">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep reinforcement learning for indoor mobile robot path planning</article-title>. <source>Sensors</source> <volume>20</volume>, <fpage>5493</fpage>. <pub-id pub-id-type="doi">10.3390/s20195493</pub-id>
<pub-id pub-id-type="pmid">32992750</pub-id>
</mixed-citation>
</ref>
<ref id="B100">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Geiger</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lenz</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Urtasun</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Are we ready for autonomous driving? The kitti vision benchmark suite</article-title>,&#x201d; in <source>
<italic>2012 IEEE conference on computer vision and pattern recognition</italic> (IEEE)</source>, <fpage>3354</fpage>&#x2013;<lpage>3361</lpage>.</mixed-citation>
</ref>
<ref id="B101">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geyer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kassahun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mahmudi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ricou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Durgesh</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>A. S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>A2d2: Audi autonomous driving dataset</article-title>. <comment>
<italic>arXiv preprint arXiv:2004.06320</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B102">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Girdhar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ramanan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sivic</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Russell</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Actionvlad: learning spatio-temporal aggregation for action classification</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>971</fpage>&#x2013;<lpage>980</lpage>.</mixed-citation>
</ref>
<ref id="B103">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Gloor</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Pedsim: pedestrian crowd simulation</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://pedsim.silmaril.org">http://pedsim.silmaril.org</ext-link>
</comment>
<volume>5</volume>.</mixed-citation>
</ref>
<ref id="B104">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
</person-group>(<year>2024</year>). <article-title>From cognition to precognition: a future-aware framework for social navigation</article-title>. <comment>
<italic>arXiv preprint arXiv:2409.13244</italic>
</comment> .</mixed-citation>
</ref>
<ref id="B105">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gonon</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Paez-Granados</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Billard</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Reactive navigation in crowds for non-holonomic robots with convex bounding shape</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>6</volume>, <fpage>4728</fpage>&#x2013;<lpage>4735</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2021.3068660</pub-id>
</mixed-citation>
</ref>
<ref id="B106">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Group</surname>
<given-names>C. D.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Crowd dynamics</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/crowddynamics/crowddynamics">https://github.com/crowddynamics/crowddynamics</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B107">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Grzeskowiak</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gonon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Dugas</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Paez-Granados</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Nieto</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Crowd against the machine: a simulation-based benchmark tool to evaluate and compare robot capabilities to navigate a human crowd</article-title>,&#x201d; in <source>
<italic>2021 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>3879</fpage>&#x2013;<lpage>3885</lpage>.</mixed-citation>
</ref>
<ref id="B108">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Gu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Stochastic trajectory prediction <italic>via</italic> motion indeterminacy diffusion</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>17113</fpage>&#x2013;<lpage>17122</lpage>.</mixed-citation>
</ref>
<ref id="B109">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Behera</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Subramanian</surname>
<given-names>V. K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A novel vision-based tracking algorithm for a human-following mobile robot</article-title>. <source>IEEE Trans. Syst. Man, Cybern. Syst.</source> <volume>47</volume>, <fpage>1415</fpage>&#x2013;<lpage>1427</lpage>. <pub-id pub-id-type="doi">10.1109/tsmc.2016.2616343</pub-id>
</mixed-citation>
</ref>
<ref id="B110">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fei-Fei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Savarese</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alahi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Social gan: socially acceptable trajectories with generative adversarial networks</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>2255</fpage>&#x2013;<lpage>2264</lpage>.</mixed-citation>
</ref>
<ref id="B111">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Guzzi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Giusti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gambardella</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Theraulaz</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Di Caro</surname>
<given-names>G. A.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Human-friendly robot navigation in dynamic environments</article-title>,&#x201d; in <source>
<italic>2013 IEEE international conference on robotics and automation</italic> (IEEE)</source>, <fpage>423</fpage>&#x2013;<lpage>430</lpage>.</mixed-citation>
</ref>
<ref id="B112">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ha</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Recurrent world models facilitate policy evolution</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>31</volume>.</mixed-citation>
</ref>
<ref id="B113">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Habibovic</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lundgren</surname>
<given-names>V. M.</given-names>
</name>
<name>
<surname>Andersson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Klingeg&#xe5;rd</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lagstr&#xf6;m</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sirkka</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Communicating intent of automated vehicles to pedestrians</article-title>. <source>Front. Psychol.</source> <volume>9</volume>, <fpage>1336</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2018.01336</pub-id>
<pub-id pub-id-type="pmid">30131737</pub-id>
</mixed-citation>
</ref>
<ref id="B114">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hahn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Krantz</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Batra</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Parikh</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Rehg</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Where are you? Localization from embodied dialog</article-title>, <fpage>806</fpage>, <lpage>822</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2020.emnlp-main.59</pub-id>
</mixed-citation>
</ref>
<ref id="B115">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hall</surname>
<given-names>E. T.</given-names>
</name>
</person-group> (<year>1963</year>). <article-title>A system for the notation of proxemic behavior</article-title>. <source>Am. Anthropol.</source> <volume>65</volume>, <fpage>1003</fpage>&#x2013;<lpage>1026</lpage>. <pub-id pub-id-type="doi">10.1525/aa.1963.65.5.02a00020</pub-id>
</mixed-citation>
</ref>
<ref id="B116">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Hamandi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>D&#x2019;Arcy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fazli</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Deepmotion: learning to navigate like humans</article-title>,&#x201d; in <source>2019 28th IEEE international conference on robot and human interactive communication (RO-MAN)</source>. <publisher-name>IEEE</publisher-name>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>.</mixed-citation>
</ref>
<ref id="B117">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>I. H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Deep reinforcement learning for robot collision avoidance with self-state-attention and sensor fusion</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>7</volume>, <fpage>6886</fpage>&#x2013;<lpage>6893</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2022.3178791</pub-id>
</mixed-citation>
</ref>
<ref id="B118">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Reinforcement learned distributed multi-robot navigation with reciprocal velocity obstacle shaped rewards</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>7</volume>, <fpage>5896</fpage>&#x2013;<lpage>5903</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2022.3161699</pub-id>
</mixed-citation>
</ref>
<ref id="B119">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hayes</surname>
<given-names>C. F.</given-names>
</name>
<name>
<surname>R&#x103;dulescu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bargiacchi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>K&#xe4;llstr&#xf6;m</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Macfarlane</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Reymond</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A practical guide to multi-objective reinforcement learning and planning</article-title>. <source>Aut. Agents Multi-Agent Syst.</source> <volume>36</volume>, <fpage>26</fpage>. <pub-id pub-id-type="doi">10.1007/s10458-022-09552-y</pub-id>
</mixed-citation>
</ref>
<ref id="B120">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Helbing</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Molnar</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Social force model for pedestrian dynamics</article-title>. <source>Phys. Rev. E</source> <volume>51</volume>, <fpage>4282</fpage>&#x2013;<lpage>4286</lpage>. <pub-id pub-id-type="doi">10.1103/physreve.51.4282</pub-id>
<pub-id pub-id-type="pmid">9963139</pub-id>
</mixed-citation>
</ref>
<ref id="B121">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Helbing</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Buzna</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Johansson</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Werner</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Self-organized pedestrian crowd dynamics: experiments, simulations, and design solutions</article-title>. <source>Transp. Sci.</source> <volume>39</volume>, <fpage>1</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1287/trsc.1040.0108</pub-id>
</mixed-citation>
</ref>
<ref id="B122">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hirose</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Shah</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sridhar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Sacson: scalable autonomous control for social navigation</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume>, <fpage>49</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2023.3329626</pub-id>
</mixed-citation>
</ref>
<ref id="B123">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ermon</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Generative adversarial imitation learning</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>29</volume>.</mixed-citation>
</ref>
<ref id="B124">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hoeller</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wellhausen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Farshidian</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Learning a state representation and navigation in cluttered and dynamic environments</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>6</volume>, <fpage>5081</fpage>&#x2013;<lpage>5088</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2021.3068639</pub-id>
</mixed-citation>
</ref>
<ref id="B125">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rodriguez-Opazo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gould</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Vln bert: a recurrent vision-and-language bert for navigation</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>1643</fpage>&#x2013;<lpage>1653</lpage>.</mixed-citation>
</ref>
<ref id="B126">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>H. N.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Q. Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Krahenbuhl</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Joint monocular 3d vehicle detection and tracking</article-title>. <source>Proc. IEEE/CVF Int. Conf. Comput. Vis.</source>, <fpage>5390</fpage>&#x2013;<lpage>5399</lpage>.</mixed-citation>
</ref>
<ref id="B127">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Stgat: modeling spatial-temporal interactions for human trajectory prediction</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>6272</fpage>&#x2013;<lpage>6281</lpage>.</mixed-citation>
</ref>
<ref id="B128">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Driggs-Campbell</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Learning sparse interaction graphs of partially detected pedestrians for trajectory prediction</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>7</volume>, <fpage>1198</fpage>&#x2013;<lpage>1205</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2021.3138547</pub-id>
</mixed-citation>
</ref>
<ref id="B129">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huber</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Weitz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Andr&#xe9;</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Amir</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Local and global explanations of agent behavior: integrating strategy summaries with saliency maps</article-title>. <source>Artif. Intell.</source> <volume>301</volume>, <fpage>103571</fpage>. <pub-id pub-id-type="doi">10.1016/j.artint.2021.103571</pub-id>
</mixed-citation>
</ref>
<ref id="B130">
<mixed-citation publication-type="web">
<collab>Inc D</collab> (<year>2025</year>). <article-title>Velodyne simulator</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://wiki.ros.org/velodyne_simulator">https://wiki.ros.org/velodyne_simulator</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B131">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jaderberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mnih</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Czarnecki</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Schaul</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Leibo</surname>
<given-names>J. Z.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Reinforcement learning with unsupervised auxiliary tasks</article-title>. <source>arXiv Prepr. arXiv:1611</source>, <fpage>05397</fpage>.</mixed-citation>
</ref>
<ref id="B132">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ghaffari</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Social zone as a barrier function for socially-compliant robot navigation</article-title>. <source>IFAC-PapersOnLine</source> <volume>58</volume>, <fpage>157</fpage>&#x2013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.1016/j.ifacol.2025.01.173</pub-id>
</mixed-citation>
</ref>
<ref id="B133">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jaradat</surname>
<given-names>M. A. K.</given-names>
</name>
<name>
<surname>Al-Rousan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Quadan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Reinforcement based mobile robot navigation in dynamic environment</article-title>. <source>Robotics Computer-Integrated Manuf.</source> <volume>27</volume>, <fpage>135</fpage>&#x2013;<lpage>149</lpage>. <pub-id pub-id-type="doi">10.1016/j.rcim.2010.06.019</pub-id>
</mixed-citation>
</ref>
<ref id="B134">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hermans</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Leibe</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Dr-spaam: a spatial-attention and auto-regressive model for person detection in 2d range data</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>10270</fpage>&#x2013;<lpage>10277</lpage>.</mixed-citation>
</ref>
<ref id="B135">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Sakib</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Graves</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jagersand</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Mapless navigation among dynamics with social-safety-awareness: a reinforcement learning approach from 2d laser scans</article-title>,&#x201d; in <source>
<italic>2020 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>6979</fpage>&#x2013;<lpage>6985</lpage>.</mixed-citation>
</ref>
<ref id="B136">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karamouzas</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Skinner</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Guy</surname>
<given-names>S. J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Universal power law governing pedestrian interactions</article-title>. <source>Phys. Rev. Lett.</source> <volume>113</volume>, <fpage>238701</fpage>. <pub-id pub-id-type="doi">10.1103/physrevlett.113.238701</pub-id>
<pub-id pub-id-type="pmid">25526171</pub-id>
</mixed-citation>
</ref>
<ref id="B137">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karnan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Warnell</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Pirk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Toshev</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Socially compliant navigation dataset (scand): a large-scale dataset of demonstrations for social navigation</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>7</volume>, <fpage>11807</fpage>&#x2013;<lpage>11814</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2022.3184025</pub-id>
</mixed-citation>
</ref>
<ref id="B138">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>K&#xe4;stner</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Buiyan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Arena-rosnav: towards deployment of deep-reinforcement-learning-based obstacle avoidance into conventional autonomous navigation systems</article-title>,&#x201d; in <source>2021 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source>. <publisher-name>IEEE</publisher-name>, <fpage>6456</fpage>&#x2013;<lpage>6463</lpage>.</mixed-citation>
</ref>
<ref id="B139">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kato</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kanda</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ishiguro</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>May i help you? Design of human-like polite approaching behavior</article-title>. <source>Proc. Tenth Annu. ACM/IEEE Int. Conf. Human-Robot Interact.</source>, <fpage>35</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1145/2696454.2696463</pub-id>
</mixed-citation>
</ref>
<ref id="B140">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Katyal</surname>
<given-names>K. D.</given-names>
</name>
<name>
<surname>Hager</surname>
<given-names>G. D.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>C. M.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Intent-aware pedestrian prediction for adaptive crowd navigation</article-title>,&#x201d; in <source>
<italic>2020 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>3277</fpage>&#x2013;<lpage>3283</lpage>.</mixed-citation>
</ref>
<ref id="B141">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>M. A. U.</given-names>
</name>
<name>
<surname>Nazir</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pagani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mokayed</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liwicki</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stricker</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A comprehensive survey of depth completion approaches</article-title>. <source>Sensors</source> <volume>22</volume>, <fpage>6969</fpage>. <pub-id pub-id-type="doi">10.3390/s22186969</pub-id>
<pub-id pub-id-type="pmid">36146318</pub-id>
</mixed-citation>
</ref>
<ref id="B142">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pineau</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Socially adaptive path planning in human environments using inverse reinforcement learning</article-title>. <source>Int. J. Soc. Robotics</source> <volume>8</volume>, <fpage>51</fpage>&#x2013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1007/s12369-015-0310-2</pub-id>
</mixed-citation>
</ref>
<ref id="B143">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>O&#x161;ep</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Leal-Taix&#xe9;</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Eagermot: 3d multi-object tracking <italic>via</italic> sensor fusion</article-title>,&#x201d; in <source>2021 IEEE international conference on robotics and automation (ICRA)</source>. <publisher-name>IEEE</publisher-name>, <fpage>11315</fpage>&#x2013;<lpage>11321</lpage>.</mixed-citation>
</ref>
<ref id="B144">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Kirillov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mintun</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ravi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rolland</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gustafson</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>Segment anything</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>4015</fpage>&#x2013;<lpage>4026</lpage>.</mixed-citation>
</ref>
<ref id="B145">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kleinmeier</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Z&#xf6;nnchen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>G&#xf6;del</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>K&#xf6;ster</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Vadere: an open-source simulation framework to promote interdisciplinary understanding</article-title>. <source>arXiv Prepr. arXiv:1907.09520</source> <volume>4</volume>, <fpage>A21</fpage>. <pub-id pub-id-type="doi">10.17815/cd.2019.21</pub-id>
</mixed-citation>
</ref>
<ref id="B146">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koenig</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Howard</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Design and use paradigms for gazebo, an open-source multi-robot simulator</article-title>. <source>
<italic>2004 IEEE/RSJ Int. Conf. intelligent robots Syst. (IROS)(IEEE Cat. No. 04CH37566)</italic> (Ieee)</source> <volume>3</volume>, <fpage>2149</fpage>&#x2013;<lpage>2154</lpage>. <pub-id pub-id-type="doi">10.1109/iros.2004.1389727</pub-id>
</mixed-citation>
</ref>
<ref id="B147">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kolve</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mottaghi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>VanderBilt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Weihs</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Herrasti</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Ai2-thor: an interactive 3d environment for visual ai</article-title>. <source>arXiv Prepr. arXiv:1712.05474</source>.</mixed-citation>
</ref>
<ref id="B148">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Korbmacher</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tordeux</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Review of pedestrian trajectory prediction methods: comparing deep learning and knowledge-based approaches</article-title>. <source>IEEE Trans. Intelligent Transp. Syst.</source> <volume>23</volume>, <fpage>24126</fpage>&#x2013;<lpage>24144</lpage>. <pub-id pub-id-type="doi">10.1109/tits.2022.3205676</pub-id>
</mixed-citation>
</ref>
<ref id="B149">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kreiss</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep social force</article-title>. <comment>
<italic>arXiv preprint arXiv:2109.12081</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B150">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kruse</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pandey</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Alami</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kirsch</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Human-aware robot navigation: a survey</article-title>. <source>Robotics Aut. Syst.</source> <volume>61</volume>, <fpage>1726</fpage>&#x2013;<lpage>1743</lpage>. <pub-id pub-id-type="doi">10.1016/j.robot.2013.05.007</pub-id>
</mixed-citation>
</ref>
<ref id="B151">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Kulh&#xe1;nek</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Derner</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>De Bruin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Babu&#x161;ka</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Vision-based navigation using deep reinforcement learning</article-title>,&#x201d; in <source>
<italic>2019 european conference on mobile robots (ECMR)</italic> (IEEE)</source>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</mixed-citation>
</ref>
<ref id="B152">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lasota</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shah</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A survey of methods for safe human-robot interaction</article-title>. <source>Found. Trends&#xae; Robotics</source> <volume>5</volume>, <fpage>261</fpage>&#x2013;<lpage>349</lpage>. <pub-id pub-id-type="doi">10.1561/2300000052</pub-id>
</mixed-citation>
</ref>
<ref id="B153">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jeong</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Velocity range-based reward shaping technique for effective map-less navigation with lidar sensor and deep reinforcement learning</article-title>. <source>Front. Neurorobotics</source> <volume>17</volume>, <fpage>1210442</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2023.1210442</pub-id>
<pub-id pub-id-type="pmid">37744086</pub-id>
</mixed-citation>
</ref>
<ref id="B154">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Adaptive and explainable deployment of navigation skills via hierarchical deep reinforcement learning</source>. <publisher-name>IEEE International Conference on Robotics and Automation ICRA</publisher-name>, <fpage>1673</fpage>&#x2013;<lpage>1679</lpage>.</mixed-citation>
</ref>
<ref id="B155">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Leigh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pineau</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Olmedo</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Person tracking and following with 2d laser scanners</article-title>,&#x201d; in <source>
<italic>2015 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>726</fpage>&#x2013;<lpage>733</lpage>.</mixed-citation>
</ref>
<ref id="B156">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lerner</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chrysanthou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lischinski</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2007</year>). &#x201c;<article-title>Crowds by example</article-title>,&#x201d;<source>Comput. Graph. forum</source>, <volume>26</volume>. <fpage>655</fpage>&#x2013;<lpage>664</lpage>. <pub-id pub-id-type="doi">10.1111/j.1467-8659.2007.01089.x</pub-id>
</mixed-citation>
</ref>
<ref id="B157">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ge</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>T. H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Role playing learning for socially concomitant mobile robot navigation</article-title>. <source>CAAI Trans. Intell. Technol.</source> <volume>3</volume>, <fpage>49</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.1049/trit.2018.0008</pub-id>
</mixed-citation>
</ref>
<ref id="B158">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Narula</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Worrall</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nebot</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Socially aware crowd navigation with multimodal pedestrian trajectory prediction for autonomous vehicles</article-title>,&#x201d; in <source>2020 IEEE 23rd international conference on intelligent transportation systems (ITSC)</source>. <publisher-name>IEEE</publisher-name>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</mixed-citation>
</ref>
<ref id="B159">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mart&#xed;n-Mart&#xed;n</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lingelbach</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Srivastava</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Igibson 2.0: object-centric simulation for robot learning of everyday household tasks</article-title>. <source>arXiv Prepr. arXiv:2108.03272</source>.</mixed-citation>
</ref>
<ref id="B160">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Self-supervised social relation representation for human group detection</article-title>,&#x201d; in <source>European conference on computer vision</source>. <publisher-name>Springer</publisher-name>, <fpage>142</fpage>&#x2013;<lpage>159</lpage>.</mixed-citation>
</ref>
<ref id="B161">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>Z. Q.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Human-aware vision-and-language navigation: bridging simulation to reality with dynamic human interactions</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>37</volume>, <fpage>119411</fpage>&#x2013;<lpage>119442</lpage>.</mixed-citation>
</ref>
<ref id="B162">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Sathyamoorthy</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Crowd-steer: realtime smooth and collision-free robot navigation in densely crowded scenarios trained using high-fidelity simulation</article-title>,&#x201d; in <source>Proceedings of the twenty-ninth international conference on international joint conferences on artificial intelligence</source>, <fpage>4221</fpage>&#x2013;<lpage>4228</lpage>.</mixed-citation>
</ref>
<ref id="B163">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>T. Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hariharan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Belongie</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Feature pyramid networks for object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>2117</fpage>&#x2013;<lpage>2125</lpage>.</mixed-citation>
</ref>
<ref id="B164">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Linh</surname>
<given-names>K. U.</given-names>
</name>
<name>
<surname>Cox</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Buiyan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lambrecht</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>All-in-one: a drl-based control switch combining state-of-the-art navigation planners</article-title>,&#x201d; in <source>2022 International Conference on Robotics and Automation (ICRA)</source>, <fpage>2861</fpage>&#x2013;<lpage>2867</lpage>. <pub-id pub-id-type="doi">10.1109/icra46639.2022.9811797</pub-id>
</mixed-citation>
</ref>
<ref id="B165">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Lisotto</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Coscia</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ballan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Social and scene-aware trajectory prediction in crowded spaces</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision workshops</source>.</mixed-citation>
</ref>
<ref id="B166">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Anguelov</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Erhan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Szegedy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Reed</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>C. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Ssd: single shot multibox detector</article-title>,&#x201d; in <source>Computer Vision&#x2013;ECCV 2016: 14th European conference, Amsterdam, the Netherlands, October 11&#x2013;14, 2016, Proceedings, part I 14</source>. <publisher-name>Springer</publisher-name>, <fpage>21</fpage>&#x2013;<lpage>37</lpage>.</mixed-citation>
</ref>
<ref id="B167">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Map-based deep imitation learning for obstacle avoidance</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>8644</fpage>&#x2013;<lpage>8649</lpage>.</mixed-citation>
</ref>
<ref id="B168">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dugas</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cesari</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Dub&#xe9;</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020a</year>). &#x201c;<article-title>Robot navigation in crowded environments using deep reinforcement learning</article-title>,&#x201d; in <source>
<italic>2020 IEEE/RSJ international conference on intelligent robots and systems (IROS)</italic> (IEEE)</source>, <fpage>5671</fpage>&#x2013;<lpage>5677</lpage>.</mixed-citation>
</ref>
<ref id="B169">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Suo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qiao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2020b</year>). <article-title>Deep learning-based localization and perception systems: approaches for autonomous cargo transportation vehicles in large-scale, semiclosed environments</article-title>. <source>IEEE Robotics and Automation Mag.</source> <volume>27</volume>, <fpage>139</fpage>&#x2013;<lpage>150</lpage>. <pub-id pub-id-type="doi">10.1109/mra.2020.2977290</pub-id>
</mixed-citation>
</ref>
<ref id="B170">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Driggs-Campbell</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Decentralized structural-rnn for robot crowd navigation with deep reinforcement learning</article-title>,&#x201d; in <source>
<italic>2021 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>3517</fpage>&#x2013;<lpage>3524</lpage>.</mixed-citation>
</ref>
<ref id="B171">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Graph relational reinforcement learning for mobile robot navigation in large-scale crowded environments</article-title>. <source>IEEE Trans. Intelligent Transp. Syst.</source> <volume>24</volume>, <fpage>8776</fpage>&#x2013;<lpage>8787</lpage>. <pub-id pub-id-type="doi">10.1109/tits.2023.3269533</pub-id>
</mixed-citation>
</ref>
<ref id="B172">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2023b</year>). &#x201c;<article-title>Intention aware robot crowd navigation with attention-based interaction graph</article-title>,&#x201d; in <source>IEEE international conference on robotics and automation (ICRA)</source>. <publisher-name>IEEE</publisher-name>, <fpage>12015</fpage>&#x2013;<lpage>12021</lpage>.</mixed-citation>
</ref>
<ref id="B173">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Y. J.</given-names>
</name>
</person-group> (<year>2023c</year>). <article-title>Visual instruction tuning</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>36</volume>, <fpage>34892</fpage>&#x2013;<lpage>34916</lpage>.</mixed-citation>
</ref>
<ref id="B174">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lerch</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Palmieri</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rudenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Koch</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ropinski</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Context-aware human behavior prediction using multimodal large language models: challenges and insights</article-title>. <source>arXiv Prepr. arXiv:2504.00839</source>.</mixed-citation>
</ref>
<ref id="B175">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Deep-learned collision avoidance policy for distributed multiagent navigation</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>2</volume>, <fpage>656</fpage>&#x2013;<lpage>663</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2017.2651371</pub-id>
</mixed-citation>
</ref>
<ref id="B176">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Towards optimally decentralized multi-robot collision avoidance <italic>via</italic> deep reinforcement learning</article-title>,&#x201d; in <source>
<italic>2018 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>6252</fpage>&#x2013;<lpage>6259</lpage>.</mixed-citation>
</ref>
<ref id="B177">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lopez</surname>
<given-names>N. G.</given-names>
</name>
<name>
<surname>Nuin</surname>
<given-names>Y. L. E.</given-names>
</name>
<name>
<surname>Moral</surname>
<given-names>E. B.</given-names>
</name>
<name>
<surname>Juan</surname>
<given-names>L. U. S.</given-names>
</name>
<name>
<surname>Rueda</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Vilches</surname>
<given-names>V. M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>gym-gazebo2, a toolkit for reinforcement learning using ros 2 and gazebo</article-title>. <source>arXiv Prepr. arXiv:1903</source>, <fpage>06278</fpage>.</mixed-citation>
</ref>
<ref id="B178">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Marshall</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Saupe</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Transalnet: towards perceptually relevant visual saliency prediction</article-title>. <source>Neurocomputing</source> <volume>494</volume>, <fpage>455</fpage>&#x2013;<lpage>467</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2022.04.080</pub-id>
</mixed-citation>
</ref>
<ref id="B179">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lowrey</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rajeswaran</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kakade</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Todorov</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mordatch</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Plan online, learn offline: efficient learning and exploration <italic>via</italic> model-based control</article-title>. <comment>
<italic>arXiv preprint arXiv:1811.01848</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B180">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Gson: a group-based social navigation framework with large multimodal model</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>10</volume>, <fpage>9646</fpage>&#x2013;<lpage>9653</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2025.3595038</pub-id>
</mixed-citation>
</ref>
<ref id="B181">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xfc;tjens</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Everett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Safe reinforcement learning with model uncertainty estimates</article-title>. <source>
<italic>Int. Conf. Robotics Automation (ICRA)</italic> (IEEE)</source>, <fpage>8662</fpage>&#x2013;<lpage>8668</lpage>. <pub-id pub-id-type="doi">10.1109/icra.2019.8793611</pub-id>
</mixed-citation>
</ref>
<ref id="B182">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Karaman</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Sparse-to-dense: depth prediction from sparse depth samples and a single image</article-title>,&#x201d; in <source>
<italic>2018 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>4796</fpage>&#x2013;<lpage>4803</lpage>.</mixed-citation>
</ref>
<ref id="B183">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Bastani</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Jayaraman</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Eureka: human-level reward design <italic>via</italic> coding large language models</article-title>. <comment>
<italic>arXiv preprint arXiv:2310.12931</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B184">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Majecka</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2009</year>). <source>Statistical models of pedestrian behaviour in the forum</source>. <publisher-loc>Citeseer</publisher-loc>: <publisher-name>Ph.D. thesis</publisher-name>.</mixed-citation>
</ref>
<ref id="B185">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Makoviychuk</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Wawrzyniak</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Storey</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Macklin</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Isaac gym: high performance gpu-based physics simulation for robot learning</article-title>. <comment>
<italic>arXiv preprint arXiv:2108.10470</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B186">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Manhardt</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kehl</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gaidon</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Roi-10d: monocular lifting of 2d detection to 6d pose and metric shape</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>2069</fpage>&#x2013;<lpage>2078</lpage>.</mixed-citation>
</ref>
<ref id="B187">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>3d object detection for autonomous driving: a comprehensive survey</article-title>. <source>Int. J. Comput. Vis.</source> <volume>131</volume>, <fpage>1909</fpage>&#x2013;<lpage>1963</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-023-01790-1</pub-id>
</mixed-citation>
</ref>
<ref id="B188">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023b</year>). &#x201c;<article-title>Leapfrog diffusion model for stochastic trajectory prediction</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>5517</fpage>&#x2013;<lpage>5526</lpage>.</mixed-citation>
</ref>
<ref id="B189">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Marta</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Holk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pek</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tumova</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Leite</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Aligning human preferences with baseline objectives in reinforcement learning</source>. <publisher-name>IEEE International Conference on Robotics and Automation ICRA</publisher-name>, <fpage>7562</fpage>&#x2013;<lpage>7568</lpage>.</mixed-citation>
</ref>
<ref id="B190">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martin-Martin</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rezatofighi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shenoi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gwak</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Frankel</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Jrdb: a dataset and benchmark of egocentric robot visual perception of humans in built environments</article-title>. <source>IEEE Trans. pattern analysis Mach. Intell.</source> <volume>45</volume>, <fpage>6748</fpage>&#x2013;<lpage>6765</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2021.3070543</pub-id>
<pub-id pub-id-type="pmid">33798067</pub-id>
</mixed-citation>
</ref>
<ref id="B191">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martinez-Baselga</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Riazuelo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Montano</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Improving robot navigation in crowded environments using intrinsic rewards</article-title>. <source>arXiv Prepr. arXiv:2302.06554</source>, <fpage>9428</fpage>&#x2013;<lpage>9434</lpage>. <pub-id pub-id-type="doi">10.1109/icra48891.2023.10160876</pub-id>
</mixed-citation>
</ref>
<ref id="B192">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Masad</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kazil</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Mesa: an agent-based modeling framework</article-title>. <source>
<italic>SciPy</italic> (Citeseer)</source>, <fpage>51</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.25080/majora-7b98e3ed-009</pub-id>
</mixed-citation>
</ref>
<ref id="B193">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matheson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Minto</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zampieri</surname>
<given-names>E. G.</given-names>
</name>
<name>
<surname>Faccio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rosati</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Human&#x2013;robot collaboration in manufacturing applications: a review</article-title>. <source>Robotics</source> <volume>8</volume>, <fpage>100</fpage>. <pub-id pub-id-type="doi">10.3390/robotics8040100</pub-id>
</mixed-citation>
</ref>
<ref id="B194">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matiisen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Oliver</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cohen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Teacher&#x2013;student curriculum learning</article-title>. <source>IEEE Trans. neural Netw. Learn. Syst.</source> <volume>31</volume>, <fpage>3732</fpage>&#x2013;<lpage>3740</lpage>. <pub-id pub-id-type="doi">10.1109/tnnls.2019.2934906</pub-id>
<pub-id pub-id-type="pmid">31502993</pub-id>
</mixed-citation>
</ref>
<ref id="B195">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mavrogiannis</surname>
<given-names>C. I.</given-names>
</name>
<name>
<surname>Thomason</surname>
<given-names>W. B.</given-names>
</name>
<name>
<surname>Knepper</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Social momentum: a framework for legible navigation in dynamic multi-agent environments</article-title>,&#x201d; in <source>Proceedings of the 2018 ACM/IEEE international conference on human-robot interaction</source>, <fpage>361</fpage>&#x2013;<lpage>369</lpage>.</mixed-citation>
</ref>
<ref id="B196">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mavrogiannis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hutchinson</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Macdonald</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Alves-Oliveira</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Knepper</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Effects of distinct robot navigation strategies on human behavior in a crowded environment</article-title>. <source>14th <italic>ACM/IEEE Int. Conf. Human-Robot Interact. (HRI)</italic> (IEEE)</source>, <fpage>421</fpage>&#x2013;<lpage>430</lpage>. <pub-id pub-id-type="doi">10.1109/hri.2019.8673115</pub-id>
</mixed-citation>
</ref>
<ref id="B197">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mavrogiannis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Baldini</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Trautman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Steinfeld</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Core challenges of social robot navigation: a survey</article-title>. <source>ACM Trans. Human-Robot Interact.</source> <volume>12</volume>, <fpage>1</fpage>&#x2013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1145/3583741</pub-id>
</mixed-citation>
</ref>
<ref id="B198">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mehta</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Diaz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Golemo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pal</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Paull</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Active domain randomization</article-title>,&#x201d; in <source>
<italic>Conference on robot learning</italic> (PMLR)</source>, <fpage>1162</fpage>&#x2013;<lpage>1176</lpage>.</mixed-citation>
</ref>
<ref id="B199">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michel</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Cyberbotics ltd. webots<sup>TM</sup>: professional mobile robot simulation</article-title>. <source>Int. J. Adv. Robotic Syst.</source> <volume>1</volume>, <fpage>5</fpage>. <pub-id pub-id-type="doi">10.5772/5618</pub-id>
</mixed-citation>
</ref>
<ref id="B200">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Milioto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vizzo</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Behley</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stachniss</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Rangenet&#x2b;&#x2b;: fast and accurate lidar semantic segmentation</article-title>,&#x201d; in <source>
<italic>2019 IEEE/RSJ international conference on intelligent robots and systems (IROS)</italic> (IEEE)</source>, <fpage>4213</fpage>&#x2013;<lpage>4220</lpage>.</mixed-citation>
</ref>
<ref id="B201">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Miller</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hasfura</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S. Y.</given-names>
</name>
<name>
<surname>How</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Dynamic arrival rate estimation for campus mobility on demand network graphs</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>2285</fpage>&#x2013;<lpage>2292</lpage>.</mixed-citation>
</ref>
<ref id="B202">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mirowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Pascanu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Viola</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Soyer</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ballard</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Banino</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Learning to navigate in complex environments</article-title>. <source>arXiv Prepr. arXiv:1611.03673</source>.</mixed-citation>
</ref>
<ref id="B203">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mirsky</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hart</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Prevention and resolution of conflicts in social navigation&#x2013;a survey</article-title>. <comment>
<italic>arXiv preprint arXiv:2106.12113</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B204">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mittal</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rudin</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hoeller</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Orbit: a unified simulation framework for interactive robot learning environments</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>8</volume>, <fpage>3740</fpage>&#x2013;<lpage>3747</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2023.3270034</pub-id>
</mixed-citation>
</ref>
<ref id="B205">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mohamed</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Elhoseiny</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Claudel</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Social-stgcnn: a social spatio-temporal graph convolutional neural network for human trajectory prediction</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>14424</fpage>&#x2013;<lpage>14432</lpage>.</mixed-citation>
</ref>
<ref id="B206">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohanan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Salgoankar</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A survey of robotic motion planning in dynamic environments</article-title>. <source>Robotics Aut. Syst.</source> <volume>100</volume>, <fpage>171</fpage>&#x2013;<lpage>185</lpage>. <pub-id pub-id-type="doi">10.1016/j.robot.2017.10.011</pub-id>
</mixed-citation>
</ref>
<ref id="B207">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>M&#xf6;ller</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Furnari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Battiato</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>H&#xe4;rm&#xe4;</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Farinella</surname>
<given-names>G. M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A survey on human-aware robot navigation</article-title>. <source>Robotics Aut. Syst.</source> <volume>145</volume>, <fpage>103837</fpage>. <pub-id pub-id-type="doi">10.1016/j.robot.2021.103837</pub-id>
</mixed-citation>
</ref>
<ref id="B208">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Monaci</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Aractingi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Silander</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Dipcan: distilling privileged information for crowd-aware navigation</article-title>. <source>Robotics Sci. Syst.</source>
</mixed-citation>
</ref>
<ref id="B209">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mousavian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Anguelov</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Flynn</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kosecka</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>3d bounding box estimation using deep learning and geometry</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>7074</fpage>&#x2013;<lpage>7082</lpage>.</mixed-citation>
</ref>
<ref id="B210">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moussa&#xef;d</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Perozo</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Garnier</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Helbing</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Theraulaz</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>The walking behaviour of pedestrian social groups and its impact on crowd dynamics</article-title>. <source>PloS one</source> <volume>5</volume>, <fpage>e10047</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0010047</pub-id>
<pub-id pub-id-type="pmid">20383280</pub-id>
</mixed-citation>
</ref>
<ref id="B211">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Munje</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Socialnav-sub: benchmarking vlms for scene understanding in social robot navigation</article-title>. <source>arXiv Prepr. arXiv:2509.08757</source>.</mixed-citation>
</ref>
<ref id="B212">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muratore</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ramos</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Turk</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gienger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Robot learning from randomized simulations: a review</article-title>. <source>Front. Robotics AI</source> <volume>9</volume>, <fpage>799893</fpage>. <pub-id pub-id-type="doi">10.3389/frobt.2022.799893</pub-id>
<pub-id pub-id-type="pmid">35494543</pub-id>
</mixed-citation>
</ref>
<ref id="B213">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Narang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Best</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Curtis</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Generating pedestrian trajectories consistent with the fundamental diagram based on physiological and psychological factors</article-title>. <source>PLoS one</source> <volume>10</volume>, <fpage>e0117856</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0117856</pub-id>
<pub-id pub-id-type="pmid">25875932</pub-id>
</mixed-citation>
</ref>
<ref id="B214">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Narasimhan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nejat</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2025</year>). &#x201c;<article-title>Olivia-nav: an online lifelong vision language approach for mobile robot social navigation</article-title>,&#x201d; in <source>
<italic>2025 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>9130</fpage>&#x2013;<lpage>9137</lpage>.</mixed-citation>
</ref>
<ref id="B215">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Narayanan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Manoghar</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Dorbala</surname>
<given-names>V. S.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bera</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Proxemo: gait-based emotion learning and multi-view proxemic fusion for socially-aware robot navigation</article-title>,&#x201d; in <source>2020 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source>. <publisher-name>IEEE</publisher-name>, <fpage>8200</fpage>&#x2013;<lpage>8207</lpage>.</mixed-citation>
</ref>
<ref id="B216">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Narvekar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sinapov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Autonomous task sequencing for customized curriculum design in reinforcement learning</source>, <source>IJCAI</source> <fpage>2536</fpage>&#x2013;<lpage>2542</lpage>. <pub-id pub-id-type="doi">10.24963/ijcai.2017/353</pub-id>
</mixed-citation>
</ref>
<ref id="B217">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Narvekar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Leonetti</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sinapov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Curriculum learning for reinforcement learning domains: a framework and survey</article-title>. <source>J. Mach. Learn. Res.</source> <volume>21</volume>, <fpage>1</fpage>&#x2013;<lpage>50</lpage>.<pub-id pub-id-type="pmid">34305477</pub-id>
</mixed-citation>
</ref>
<ref id="B218">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naseer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Porikli</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Indoor scene understanding in 2.5/3d for autonomous agents: a survey</article-title>. <source>IEEE access</source> <volume>7</volume>, <fpage>1859</fpage>&#x2013;<lpage>1887</lpage>. <pub-id pub-id-type="doi">10.1109/access.2018.2886133</pub-id>
</mixed-citation>
</ref>
<ref id="B219">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Nazeri</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Payandeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Datar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Toward human-like social robot navigation: a large-scale, multi-modal, social human navigation dataset</article-title>,&#x201d; in <source>
<italic>2023 IEEE/RSJ international conference on intelligent robots and systems (IROS)</italic> (IEEE)</source>, <fpage>7442</fpage>&#x2013;<lpage>7447</lpage>.</mixed-citation>
</ref>
<ref id="B220">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Nishimura</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yonetani</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <source>L2b: learning to balance the safety-efficiency trade-off in interactive crowd-aware robot navigation</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>11004</fpage>&#x2013;<lpage>11010</lpage>.</mixed-citation>
</ref>
<ref id="B221">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hoogs</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Perera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cuntoor</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J. T.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). &#x201c;<article-title>A large-scale benchmark dataset for event recognition in surveillance video</article-title>,&#x201d;<source>CVPR</source>, <fpage>3153</fpage>&#x2013;<lpage>3160</lpage>.</mixed-citation>
</ref>
<ref id="B222">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Value prediction network</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>30</volume>.</mixed-citation>
</ref>
<ref id="B223">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Okal</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Arras</surname>
<given-names>K. O.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Learning socially normative robot navigation behaviors with bayesian inverse reinforcement learning</article-title>,&#x201d; in <source>
<italic>2016 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>2889</fpage>&#x2013;<lpage>2895</lpage>.</mixed-citation>
</ref>
<ref id="B224">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Okunevich</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Lombard</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Krajnik</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ruichek</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Online context learning for socially compliant navigation</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>10</volume>, <fpage>5042</fpage>&#x2013;<lpage>5049</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2025.3557309</pub-id>
</mixed-citation>
</ref>
<ref id="B225">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ouyang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Almeida</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wainwright</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mishkin</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Training language models to follow instructions with human feedback</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>35</volume>, <fpage>27730</fpage>&#x2013;<lpage>27744</lpage>.</mixed-citation>
</ref>
<ref id="B226">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paez-Granados</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gonon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Huber</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Billard</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>3d point cloud and rgbd of pedestrians in robot crowd navigation: detection and tracking</article-title>. <source>IEEE DataPort</source> <volume>12</volume>.</mixed-citation>
</ref>
<ref id="B227">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Paez-Granados</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gonon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Leibe</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Suzuki</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <source>Pedestrian-robot interactions on autonomous crowd navigation: reactive control methods and evaluation metrics</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>149</fpage>&#x2013;<lpage>156</lpage>.</mixed-citation>
</ref>
<ref id="B228">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Simpletrack: understanding and rethinking 3d multi-object tracking</article-title>,&#x201d; in <source>European conference on computer vision</source>. <publisher-name>Springer</publisher-name>, <fpage>680</fpage>&#x2013;<lpage>696</lpage>.</mixed-citation>
</ref>
<ref id="B229">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parker-Holder</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rajan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Biedenkapp</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Eimer</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Automated reinforcement learning (autorl): a survey and open problems</article-title>. <source>J. Artif. Intell. Res.</source> <volume>74</volume>, <fpage>517</fpage>&#x2013;<lpage>568</lpage>. <pub-id pub-id-type="doi">10.1613/jair.1.13596</pub-id>
</mixed-citation>
</ref>
<ref id="B230">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pathak</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Efros</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Darrell</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Curiosity-driven exploration by self-supervised prediction</article-title>,&#x201d; in <source>International Conference on Machine Learning</source>. <publisher-loc>Sydney, Australia</publisher-loc>: <publisher-name>PMLR</publisher-name>, <fpage>2778</fpage>&#x2013;<lpage>2787</lpage>.</mixed-citation>
</ref>
<ref id="B231">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Paxton</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Raman</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Hager</surname>
<given-names>G. D.</given-names>
</name>
<name>
<surname>Kobilarov</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Combining neural networks and tree search for task and motion planning in challenging environments</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>6059</fpage>&#x2013;<lpage>6066</lpage>.</mixed-citation>
</ref>
<ref id="B232">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Payandeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nazeri</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mukherjee</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Raj</surname>
<given-names>A. H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Social-llava: enhancing robot navigation through human-language reasoning in social spaces</article-title>. <source>arXiv Prepr. arXiv:2501.09024</source>.</mixed-citation>
</ref>
<ref id="B233">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pellegrini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ess</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schindler</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Van Gool</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>You&#x2019;ll never walk alone: modeling social behavior for multi-target tracking</article-title>,&#x201d; in <source>
<italic>2009 IEEE 12th international conference on computer vision</italic> (IEEE)</source>, <fpage>261</fpage>&#x2013;<lpage>268</lpage>.</mixed-citation>
</ref>
<ref id="B234">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>3d multi-object tracking in autonomous driving: a survey</article-title>,&#x201d; in <source>2024 36th Chinese control and decision conference (CCDC)</source>. <publisher-name>IEEE</publisher-name>, <fpage>4964</fpage>&#x2013;<lpage>4971</lpage>.</mixed-citation>
</ref>
<ref id="B235">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pfeiffer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schaeuble</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nieto</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cadena</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>From perception to decision: a data-driven approach to end-to-end motion planning for autonomous ground robots</article-title>,&#x201d; in <source>
<italic>2017 ieee international conference on robotics and automation (icra)</italic> (IEEE)</source>, <fpage>1527</fpage>&#x2013;<lpage>1533</lpage>.</mixed-citation>
</ref>
<ref id="B236">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pfeiffer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shukla</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Turchetta</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cadena</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Krause</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Reinforced imitation: sample efficient deep reinforcement learning for mapless navigation by leveraging prior demonstrations</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>3</volume>, <fpage>4423</fpage>&#x2013;<lpage>4430</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2018.2869644</pub-id>
</mixed-citation>
</ref>
<ref id="B237">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pinto</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Andrychowicz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Welinder</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zaremba</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Asymmetric actor critic for image-based robot learning</article-title>. <comment>
<italic>arXiv preprint arXiv:1710.06542</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B238">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pirk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Takayama</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Francis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Toshev</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A protocol for validating social navigation policies</article-title>. <comment>
<italic>arXiv preprint arXiv:2204.05443</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B239">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Poddar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mavrogiannis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Srinivasa</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2023</year>). <source>From crowd motion prediction to robot navigation in crowds</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>6765</fpage>&#x2013;<lpage>6772</lpage>.</mixed-citation>
</ref>
<ref id="B240">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pramanik</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pal</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Maiti</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mitra</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Granulated rcnn and multi-class deep sort for multi-object detection and tracking</article-title>. <source>IEEE Trans. Emerg. Top. Comput. Intell.</source> <volume>6</volume>, <fpage>171</fpage>&#x2013;<lpage>181</lpage>. <pub-id pub-id-type="doi">10.1109/tetci.2020.3041019</pub-id>
</mixed-citation>
</ref>
<ref id="B241">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Puig</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Undersander</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Szot</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cote</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>T. Y.</given-names>
</name>
<name>
<surname>Partsey</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Habitat 3.0: a co-habitat for humans, avatars and robots</article-title>. <comment>
<italic>arXiv preprint arXiv:2310.13724</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B242">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guibas</surname>
<given-names>L. J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Pointnet&#x2b;&#x2b;: deep hierarchical feature learning on point sets in a metric space</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>30</volume>.</mixed-citation>
</ref>
<ref id="B243">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guibas</surname>
<given-names>L. J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Frustum pointnets for 3d object detection from rgb-d data</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>918</fpage>&#x2013;<lpage>927</lpage>.</mixed-citation>
</ref>
<ref id="B244">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Monogrnet: a geometric reasoning network for monocular 3d object localization</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>33</volume>, <fpage>8851</fpage>&#x2013;<lpage>8858</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v33i01.33018851</pub-id>
</mixed-citation>
</ref>
<ref id="B245">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Qin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rus</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Deep imitation learning for autonomous navigation in dynamic pedestrian environments</source>. <publisher-name>IEEE International Conference on Robotics and Automation ICRA</publisher-name>, <fpage>4108</fpage>&#x2013;<lpage>4115</lpage>.</mixed-citation>
</ref>
<ref id="B246">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qiao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>T. S.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>Unrealcv: virtual worlds for computer vision</article-title>,&#x201d; in <source>Proceedings of the 25th ACM international conference on multimedia</source>, <fpage>1221</fpage>&#x2013;<lpage>1224</lpage>.</mixed-citation>
</ref>
<ref id="B247">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Qu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Llms are good action recognizers</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>18395</fpage>&#x2013;<lpage>18406</lpage>.</mixed-citation>
</ref>
<ref id="B248">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rakai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Data association in multiple object tracking: a survey of recent techniques</article-title>. <source>Expert Syst. Appl.</source> <volume>192</volume>, <fpage>116300</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2021.116300</pub-id>
</mixed-citation>
</ref>
<ref id="B249">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>You only look once: unified, real-time object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>.</mixed-citation>
</ref>
<ref id="B250">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Yolov3: an incremental improvement</article-title>. <comment>
<italic>arXiv preprint arXiv:1804.02767</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B251">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Faster r-cnn: towards real-time object detection with region proposal networks</article-title>. <source>IEEE Trans. pattern analysis Mach. Intell.</source> <volume>39</volume>, <fpage>1137</fpage>&#x2013;<lpage>1149</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2016.2577031</pub-id>
<pub-id pub-id-type="pmid">27295650</pub-id>
</mixed-citation>
</ref>
<ref id="B252">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Repiso</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Garrell</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sanfeliu</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>People&#x2019;s adaptive side-by-side model evolved to accompany groups of people by social robots</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>2387</fpage>&#x2013;<lpage>2394</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2020.2970676</pub-id>
</mixed-citation>
</ref>
<ref id="B253">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ridel</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Deo</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wolf</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Trivedi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Scene compliant trajectory forecast with agent-centric spatio-temporal grids</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>2816</fpage>&#x2013;<lpage>2823</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2020.2974393</pub-id>
</mixed-citation>
</ref>
<ref id="B254">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Riedmiller</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hafner</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lampe</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Neunert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Degrave</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wiele</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Learning by playing solving sparse reward tasks from scratch</article-title>,&#x201d; in <source>
<italic>International conference on machine learning</italic> (PMLR)</source>, <fpage>4344</fpage>&#x2013;<lpage>4353</lpage>.</mixed-citation>
</ref>
<ref id="B255">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rios-Martinez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Spalanzani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Laugier</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>From proxemics theory to socially-aware navigation: a survey</article-title>. <source>Int. J. Soc. Robotics</source> <volume>7</volume>, <fpage>137</fpage>&#x2013;<lpage>153</lpage>. <pub-id pub-id-type="doi">10.1007/s12369-014-0251-1</pub-id>
</mixed-citation>
</ref>
<ref id="B256">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Robicquet</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sadeghian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alahi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Savarese</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Learning social etiquette: human trajectory understanding in crowded scenes</article-title>,&#x201d; in <source>Computer Vision&#x2013;ECCV 2016: 14th European conference, Amsterdam, the Netherlands, October 11-14, 2016, proceedings, part VIII 14</source>. <publisher-name>Springer</publisher-name>, <fpage>549</fpage>&#x2013;<lpage>565</lpage>.</mixed-citation>
</ref>
<ref id="B257">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roijers</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Vamplew</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Whiteson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dazeley</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A survey of multi-objective sequential decision-making</article-title>. <source>J. Artif. Intell. Res.</source> <volume>48</volume>, <fpage>67</fpage>&#x2013;<lpage>113</lpage>. <pub-id pub-id-type="doi">10.1613/jair.3987</pub-id>
</mixed-citation>
</ref>
<ref id="B258">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>R&#xf6;smann</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hoffmann</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bertram</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Timed-elastic-bands for time-optimal point-to-point nonlinear model predictive control</article-title>,&#x201d; in <source>2015 european control conference (ECC)</source>. <publisher-name>IEEE</publisher-name>, <fpage>3352</fpage>&#x2013;<lpage>3357</lpage>.</mixed-citation>
</ref>
<ref id="B259">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Ross</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bagnell</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>A reduction of imitation learning and structured prediction to no-regret online learning</article-title>,&#x201d; in <source>Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics</source>. <publisher-loc>Fort Lauderdale, FL, United States</publisher-loc>: <publisher-name>JMLR Workshop and Conference Proceedings</publisher-name>, <fpage>627</fpage>&#x2013;<lpage>635</lpage>.</mixed-citation>
</ref>
<ref id="B260">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Roth</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Xai-n: sensor-based robot navigation using expert policies and decision trees</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>2053</fpage>&#x2013;<lpage>2060</lpage>.</mixed-citation>
</ref>
<ref id="B261">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Roth</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Nubert</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mittal</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Viplanner: visual semantic imperative learning for local navigation</article-title>,&#x201d; in <source>
<italic>2024 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>5243</fpage>&#x2013;<lpage>5249</lpage>.</mixed-citation>
</ref>
<ref id="B262">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rudenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Palmieri</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Herman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kitani</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Gavrila</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Arras</surname>
<given-names>K. O.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Human motion trajectory prediction: a survey</article-title>. <source>Int. J. Robotics Res.</source> <volume>39</volume>, <fpage>895</fpage>&#x2013;<lpage>935</lpage>. <pub-id pub-id-type="doi">10.1177/0278364920917446</pub-id>
</mixed-citation>
</ref>
<ref id="B263">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rudenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kucner</surname>
<given-names>T. P.</given-names>
</name>
<name>
<surname>Swaminathan</surname>
<given-names>C. S.</given-names>
</name>
<name>
<surname>Chadalavada</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Arras</surname>
<given-names>K. O.</given-names>
</name>
<name>
<surname>Lilienthal</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>TH&#xd6;R: human-robot navigation data collection and accurate motion trajectories dataset</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>676</fpage>&#x2013;<lpage>682</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2020.2965416</pub-id>
</mixed-citation>
</ref>
<ref id="B264">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rusu</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Colmenarejo</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Gulcehre</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Desjardins</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kirkpatrick</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pascanu</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Policy distillation</article-title>. <comment>
<italic>arXiv preprint arXiv:1511.06295</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B265">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Sadeghian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kosaraju</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Sadeghian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hirose</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Rezatofighi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Savarese</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Sophie: an attentive gan for predicting paths compliant to social and physical constraints</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>1349</fpage>&#x2013;<lpage>1358</lpage>.</mixed-citation>
</ref>
<ref id="B266">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Samsani</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Muhammad</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Socially compliant robot navigation in crowded environment by human behavior resemblance using deep reinforcement learning</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>6</volume>, <fpage>5223</fpage>&#x2013;<lpage>5230</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2021.3071954</pub-id>
</mixed-citation>
</ref>
<ref id="B267">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>S&#xe1;nchez-Ib&#xe1;&#xf1;ez</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>P&#xe9;rez-del Pulgar</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Garc&#xed;a-Cerezo</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Path planning for autonomous mobile robots: a review</article-title>. <source>Sensors</source> <volume>21</volume>, <fpage>7898</fpage>. <pub-id pub-id-type="doi">10.3390/s21237898</pub-id>
<pub-id pub-id-type="pmid">34883899</pub-id>
</mixed-citation>
</ref>
<ref id="B268">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Sathyamoorthy</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chandra</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020a</year>). <source>Densecavoid: real-time navigation in dense crowds using anticipatory behaviors</source>. <publisher-name>IEEE International Conference on Robotics and Automation ICRA</publisher-name>, <fpage>11345</fpage>&#x2013;<lpage>11352</lpage>.</mixed-citation>
</ref>
<ref id="B269">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sathyamoorthy</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Frozone: freezing-free, pedestrian-friendly navigation in human crowds</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>4352</fpage>&#x2013;<lpage>4359</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2020.2996593</pub-id>
</mixed-citation>
</ref>
<ref id="B270">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Savva</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kadian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maksymets</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wijmans</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>Habitat: a platform for embodied ai research</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>9339</fpage>&#x2013;<lpage>9347</lpage>.</mixed-citation>
</ref>
<ref id="B271">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klimov</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Proximal policy optimization algorithms</article-title>. <comment>
<italic>arXiv preprint arXiv:1707.06347</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B272">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Seitz</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>K&#xf6;ster</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Natural discretization of pedestrian movement in continuous space</article-title>. <source>Phys. Rev. E&#x2014;Statistical, Nonlinear, Soft Matter Phys.</source> <volume>86</volume>, <fpage>046108</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevE.86.046108</pub-id>
<pub-id pub-id-type="pmid">23214653</pub-id>
</mixed-citation>
</ref>
<ref id="B273">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>K. S.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>End-to-end navigation strategy with deep reinforcement learning for mobile robots</article-title>. <source>IEEE Trans. Industrial Inf.</source> <volume>16</volume>, <fpage>2393</fpage>&#x2013;<lpage>2402</lpage>. <pub-id pub-id-type="doi">10.1109/tii.2019.2936167</pub-id>
</mixed-citation>
</ref>
<ref id="B274">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019b</year>). &#x201c;<article-title>Pointrcnn: 3d object proposal generation and detection from point cloud</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>770</fpage>&#x2013;<lpage>779</lpage>.</mixed-citation>
</ref>
<ref id="B275">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sighencea</surname>
<given-names>B. I.</given-names>
</name>
<name>
<surname>Stanciu</surname>
<given-names>R. I.</given-names>
</name>
<name>
<surname>C&#x103;leanu</surname>
<given-names>C. D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A review of deep learning-based methods for pedestrian trajectory prediction</article-title>. <source>Sensors</source> <volume>21</volume>, <fpage>7543</fpage>. <pub-id pub-id-type="doi">10.3390/s21227543</pub-id>
<pub-id pub-id-type="pmid">34833619</pub-id>
</mixed-citation>
</ref>
<ref id="B276">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Two-stream convolutional networks for action recognition in videos</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>27</volume>.</mixed-citation>
</ref>
<ref id="B277">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Singamaneni</surname>
<given-names>P. T.</given-names>
</name>
<name>
<surname>Favier</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alami</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Watch out! there may be a human. addressing invisible humans in social navigation</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>11344</fpage>&#x2013;<lpage>11351</lpage>.</mixed-citation>
</ref>
<ref id="B278">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singamaneni</surname>
<given-names>P. T.</given-names>
</name>
<name>
<surname>Bachiller-Burgos</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Manso</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Garrell</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sanfeliu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Spalanzani</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>A survey on socially aware robot navigation: taxonomy and future challenges</article-title>. <source>Int. J. Robotics Res.</source>, <fpage>02783649241230562</fpage>.</mixed-citation>
</ref>
<ref id="B279">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smart</surname>
<given-names>W. D.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Practical reinforcement learning in continuous spaces</article-title>. <source>ICML</source>, <fpage>903</fpage>&#x2013;<lpage>910</lpage>.</mixed-citation>
</ref>
<ref id="B280">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smart</surname>
<given-names>W. D.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
</person-group> (<year>2002</year>). &#x201c;<article-title>Effective reinforcement learning for mobile robots</article-title>,&#x201d; in <source>Proceedings 2002 IEEE international conference on robotics and automation (cat. No. 02CH37292)</source>, <source>IEEE</source> <volume>4</volume>, <fpage>3404</fpage>&#x2013;<lpage>3410</lpage>. <pub-id pub-id-type="doi">10.1109/robot.2002.1014237</pub-id>
</mixed-citation>
</ref>
<ref id="B281">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Payandeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Raj</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Vlm-social-nav: socially aware robot navigation through scoring using vision-language models</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>10</volume>, <fpage>508</fpage>&#x2013;<lpage>515</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2024.3511409</pub-id>
</mixed-citation>
</ref>
<ref id="B282">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sprague</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chandra</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Holtz</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Biswas</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Socialgym 2.0: simulator for multi-agent social robot navigation in shared human spaces</article-title>. <source>arXiv Prepr. arXiv:2303.05584</source>.</mixed-citation>
</ref>
<ref id="B283">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stratton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hauser</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mavrogiannis</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Characterizing the complexity of social robot navigation scenarios</article-title>. <source>arXiv Prepr. arXiv:2405.11410</source>.</mixed-citation>
</ref>
<ref id="B284">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Strigel</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Meissner</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Seeliger</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wilking</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Dietmayer</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>The ko-per intersection laserscanner and video dataset</article-title>,&#x201d; in <source>17th international IEEE conference on intelligent transportation systems (ITSC)</source>. <publisher-name>IEEE</publisher-name>, <fpage>1900</fpage>&#x2013;<lpage>1901</lpage>.</mixed-citation>
</ref>
<ref id="B285">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>St&#xfc;vel</surname>
<given-names>S. A.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Python-rvo2 library</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/sybrenstuvel/Python-RVO2">https://github.com/sybrenstuvel/Python-RVO2</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B286">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Crowd navigation in an unknown and dynamic environment based on deep reinforcement learning</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>109544</fpage>&#x2013;<lpage>109554</lpage>. <pub-id pub-id-type="doi">10.1109/access.2019.2933492</pub-id>
</mixed-citation>
</ref>
<ref id="B287">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Elsayed</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bewley</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Rsn: range sparse net for efficient, accurate lidar 3d object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>5725</fpage>&#x2013;<lpage>5734</lpage>.</mixed-citation>
</ref>
<ref id="B288">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Szot</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Clegg</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Undersander</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wijmans</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Turner</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Habitat 2.0: training home assistants to rearrange their habitat</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>34</volume>, <fpage>251</fpage>&#x2013;<lpage>266</lpage>.</mixed-citation>
</ref>
<ref id="B289">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Paolo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Virtual-to-real deep reinforcement learning: continuous control of mobile robots for mapless navigation</article-title>,&#x201d; in <source>2017 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source>. <publisher-name>IEEE</publisher-name>, <fpage>31</fpage>&#x2013;<lpage>36</lpage>.</mixed-citation>
</ref>
<ref id="B290">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Socially compliant navigation through raw depth inputs with generative adversarial imitation learning</article-title>,&#x201d; in <source>2018 IEEE international conference on robotics and automation (ICRA)</source>. <publisher-name>IEEE</publisher-name>, <fpage>1111</fpage>&#x2013;<lpage>1117</lpage>.</mixed-citation>
</ref>
<ref id="B291">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tamar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Value iteration networks</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>29</volume>.</mixed-citation>
</ref>
<ref id="B292">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Deepmnavigate: deep reinforced multi-robot navigation unifying local and global collision avoidance</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS IEEE</publisher-name>, <fpage>6952</fpage>&#x2013;<lpage>6959</lpage>.</mixed-citation>
</ref>
<ref id="B293">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Thalhammer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Patten</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vincze</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kropatsch</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Sydd: synthetic depth data randomization for object detection using domain-relevant background</article-title>. <publisher-loc>Stift Vorau, Austria</publisher-loc>: <publisher-name>Computer Vision Winter Workshop</publisher-name>, <fpage>14</fpage>&#x2013;<lpage>22</lpage>.</mixed-citation>
</ref>
<ref id="B294">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thomaz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hoffman</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cakmak</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Computational human-robot interaction</article-title>. <source>Found. Trends&#xae; Robotics</source> <volume>4</volume> (<issue>2-3</issue>), <fpage>105</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1561/2300000049</pub-id>
</mixed-citation>
</ref>
<ref id="B295">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thrun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Beetz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bennewitz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Cremers</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Dellaert</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2000</year>). <article-title>Probabilistic algorithms and the interactive museum tour-guide robot minerva</article-title>. <source>Int. J. robotics Res.</source> <volume>19</volume>, <fpage>972</fpage>&#x2013;<lpage>999</lpage>. <pub-id pub-id-type="doi">10.1177/02783640022067922</pub-id>
</mixed-citation>
</ref>
<ref id="B296">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tobin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ray</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zaremba</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Domain randomization for transferring deep neural networks from simulation to the real world</source>. <publisher-name>IEEE/RSJ international conference on intelligent robots and systems IROS</publisher-name>, <fpage>23</fpage>&#x2013;<lpage>30</lpage>.</mixed-citation>
</ref>
<ref id="B297">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tongloy</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chuwongin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jaksukam</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chousangsuntorn</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Boonsang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Asynchronous deep reinforcement learning for the mobile robot navigation with supervised auxiliary tasks</article-title>,&#x201d; in <source>2017 2nd international conference on robotics and automation engineering (ICRAE)</source>. <publisher-name>IEEE</publisher-name>, <fpage>68</fpage>&#x2013;<lpage>72</lpage>.</mixed-citation>
</ref>
<ref id="B298">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Truong</surname>
<given-names>X. T.</given-names>
</name>
<name>
<surname>Ngo</surname>
<given-names>T. D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>&#x201c;to approach humans?&#x201d;: a unified framework for approaching pose prediction and socially aware robot navigation</article-title>. <source>IEEE Trans. Cognitive Dev. Syst.</source> <volume>10</volume>, <fpage>557</fpage>&#x2013;<lpage>572</lpage>. <pub-id pub-id-type="doi">10.1109/tcds.2017.2751963</pub-id>
</mixed-citation>
</ref>
<ref id="B299">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A generative approach for socially compliant navigation</article-title>,&#x201d; in <source>
<italic>2020 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>2160</fpage>&#x2013;<lpage>2166</lpage>.</mixed-citation>
</ref>
<ref id="B300">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tsoi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hussein</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Espinoza</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ruiz</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>V&#xe1;zquez</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Sean: social environment for autonomous navigation</article-title>,&#x201d; in <source>Proceedings of the 8th international conference on human-agent interaction</source>, <fpage>281</fpage>&#x2013;<lpage>283</lpage>.</mixed-citation>
</ref>
<ref id="B301">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Tsoi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hussein</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fugikawa</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>V&#xe1;zquez</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <source>An approach to deploy interactive robotic simulators on the web for hri experiments: results in social robot navigation</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>7528</fpage>&#x2013;<lpage>7535</lpage>.</mixed-citation>
</ref>
<ref id="B302">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsoi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Sohn</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Schwartz</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ramesh</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Sean 2.0: formalizing and generating social situations for robot navigation</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>7</volume>, <fpage>11047</fpage>&#x2013;<lpage>11054</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2022.3196783</pub-id>
</mixed-citation>
</ref>
<ref id="B303">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Van den Berg</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2008</year>). &#x201c;<article-title>Reciprocal velocity obstacles for real-time multi-agent navigation</article-title>,&#x201d; in <source>
<italic>2008 IEEE international conference on robotics and automation</italic> (Ieee)</source>, <fpage>1928</fpage>&#x2013;<lpage>1935</lpage>.</mixed-citation>
</ref>
<ref id="B304">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Van Den Berg</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guy</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Reciprocal n-body collision avoidance</article-title>,&#x201d; in <source>Robotics research: the 14th international symposium ISRR</source>. <publisher-name>Springer</publisher-name>, <fpage>3</fpage>&#x2013;<lpage>19</lpage>.</mixed-citation>
</ref>
<ref id="B305">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Van Den Berg</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guy</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Snape</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Rvo2 library</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://gamma.cs.unc.edu/RVO2">https://gamma.cs.unc.edu/RVO2</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B306">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>van Toll</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Grzeskowiak</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gand&#xed;a</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Amirian</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Berton</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bruneau</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Generalized microscropic crowd simulation using costs in velocity space</article-title>,&#x201d; in <source>Symposium on interactive 3D graphics and games</source>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>.</mixed-citation>
</ref>
<ref id="B307">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Vasquez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Okal</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Arras</surname>
<given-names>K. O.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Inverse reinforcement learning algorithms and features for robot navigation in crowds: an experimental comparison</article-title>,&#x201d; in <source>
<italic>2014 IEEE/RSJ international conference on intelligent robots and systems</italic> (IEEE)</source>, <fpage>1341</fpage>&#x2013;<lpage>1346</lpage>.</mixed-citation>
</ref>
<ref id="B308">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Vora</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lang</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Helou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Beijbom</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Pointpainting: sequential fusion for 3d object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>4604</fpage>&#x2013;<lpage>4612</lpage>.</mixed-citation>
</ref>
<ref id="B309">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vouros</surname>
<given-names>G. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Explainable deep reinforcement learning: state of the art and challenges</article-title>. <source>ACM Comput. Surv.</source> <volume>55</volume>, <fpage>1</fpage>&#x2013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1145/3527448</pub-id>
</mixed-citation>
</ref>
<ref id="B310">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Vuong</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Vu</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Binh</surname>
<given-names>H. T. T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>Habicrowd: a high performance simulator for crowd-aware visual navigation</article-title>,&#x201d; in <source>arXiv preprint arXiv:2306.11377</source>.</mixed-citation>
</ref>
<ref id="B311">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep visual domain adaptation: a survey</article-title>. <source>Neurocomputing</source> <volume>312</volume>, <fpage>135</fpage>&#x2013;<lpage>153</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2018.05.083</pub-id>
</mixed-citation>
</ref>
<ref id="B312">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018a</year>). <article-title>Learning to navigate through complex dynamic environment with modular deep reinforcement learning</article-title>. <source>IEEE Trans. Games</source> <volume>10</volume>, <fpage>400</fpage>&#x2013;<lpage>412</lpage>. <pub-id pub-id-type="doi">10.1109/tg.2018.2849942</pub-id>
</mixed-citation>
</ref>
<ref id="B313">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018b</year>). &#x201c;<article-title>Non-local neural networks</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>7794</fpage>&#x2013;<lpage>7803</lpage>.</mixed-citation>
</ref>
<ref id="B314">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018c</year>). <article-title>Detecting coherent groups in crowd scenes by multiview clustering</article-title>. <source>IEEE Trans. pattern analysis Mach. Intell.</source> <volume>42</volume>, <fpage>46</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2018.2875002</pub-id>
<pub-id pub-id-type="pmid">30307858</pub-id>
</mixed-citation>
</ref>
<ref id="B315">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Resilient navigation among dynamic agents with hierarchical reinforcement learning</article-title>,&#x201d; in <source>Advances in computer graphics: 38th computer graphics international conference, CGI 2021, virtual event, September 6&#x2013;10, 2021, proceedings 38</source>. <publisher-name>Springer</publisher-name>, <fpage>504</fpage>&#x2013;<lpage>516</lpage>.</mixed-citation>
</ref>
<ref id="B316">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Min</surname>
<given-names>B. C.</given-names>
</name>
</person-group> (<year>2022a</year>). &#x201c;<article-title>Feedback-efficient active preference learning for socially aware robot navigation</article-title>,&#x201d; in <source>2022 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source>. <publisher-name>IEEE</publisher-name>, <fpage>11336</fpage>&#x2013;<lpage>11343</lpage>.</mixed-citation>
</ref>
<ref id="B317">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Deepfusionmot: a 3d multi-object tracking framework based on camera-lidar fusion with deep association</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>7</volume>, <fpage>8260</fpage>&#x2013;<lpage>8267</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2022.3187264</pub-id>
</mixed-citation>
</ref>
<ref id="B318">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>W. P.</given-names>
</name>
<name>
<surname>Carreno-Medrano</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cosgun</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Croft</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022c</year>). &#x201c;<article-title>Metrics for evaluating social conformity of crowd navigation algorithms</article-title>,&#x201d; in <source>2022 IEEE international conference on advanced robotics and its social impacts (ARSO)</source>. <publisher-name>IEEE</publisher-name>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation>
</ref>
<ref id="B319">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Min</surname>
<given-names>B. C.</given-names>
</name>
</person-group> (<year>2023a</year>). <source>Navistar: socially aware robot navigation with hybrid spatio-temporal graph transformer and preference learning</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>11348</fpage>&#x2013;<lpage>11355</lpage>.</mixed-citation>
</ref>
<ref id="B320">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2023b</year>). <article-title>Camo-mot: combined appearance-motion optimization for 3d multi-object tracking with camera-lidar fusion</article-title>. <source>IEEE Trans. Intelligent Transp. Syst.</source> <volume>24</volume>, <fpage>11981</fpage>&#x2013;<lpage>11996</lpage>. <pub-id pub-id-type="doi">10.1109/tits.2023.3285651</pub-id>
</mixed-citation>
</ref>
<ref id="B321">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Obi</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Min</surname>
<given-names>B. C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Unifying large language model and deep reinforcement learning for human-in-loop interactive socially-aware navigation</article-title>. <comment>
<italic>arXiv preprint arXiv:2403.15648</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B322">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Held</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kitani</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Ab3dmot: a baseline for 3d multi-object tracking and new evaluation metrics</article-title>. <comment>
<italic>arXiv preprint arXiv:2008.08063</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B323">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wijmans</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Kadian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Morcos</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Essa</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Parikh</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Dd-ppo: learning near-perfect pointgoal navigators from 2.5 billion frames</article-title>. <comment>
<italic>arXiv preprint arXiv:1911.00357</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B324">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wijmans</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Savva</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Essa</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Morcos</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Batra</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Emergence of maps in the memories of blind navigation agents</article-title>. <source>AI Matters</source> <volume>9</volume>, <fpage>8</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1145/3609468.3609471</pub-id>
</mixed-citation>
</ref>
<ref id="B325">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wojke</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bewley</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Paulus</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Simple online and realtime tracking with a deep association metric</article-title>,&#x201d; in <source>
<italic>2017 IEEE international conference on image processing (ICIP)</italic> (IEEE)</source>, <fpage>3645</fpage>&#x2013;<lpage>3649</lpage>.</mixed-citation>
</ref>
<ref id="B326">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Vision-language navigation: a survey and taxonomy</article-title>. <source>Neural Comput. Appl.</source> <volume>36</volume>, <fpage>3291</fpage>&#x2013;<lpage>3316</lpage>. <pub-id pub-id-type="doi">10.1007/s00521-023-09217-1</pub-id>
</mixed-citation>
</ref>
<ref id="B327">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Xiang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mo</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Sapien: a simulated part-based interactive environment</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>11097</fpage>&#x2013;<lpage>11107</lpage>.</mixed-citation>
</ref>
<ref id="B328">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dames</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Drl-vo: learning to navigate through crowded dynamic scenes using velocity obstacles</article-title>. <source>IEEE Trans. Robotics</source> <volume>39</volume>, <fpage>2700</fpage>&#x2013;<lpage>2719</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2023.3257549</pub-id>
</mixed-citation>
</ref>
<ref id="B329">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rosa</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Markham</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Trigoni</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Learning with training wheels: speeding up training with a simple controller for deep reinforcement learning</article-title>,&#x201d; in <source>
<italic>2018 IEEE international conference on robotics and automation (ICRA)</italic> (IEEE)</source>, <fpage>6276</fpage>&#x2013;<lpage>6283</lpage>.</mixed-citation>
</ref>
<ref id="B330">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Multi-level fusion based 3d object detection from monocular images</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>2345</fpage>&#x2013;<lpage>2353</lpage>.</mixed-citation>
</ref>
<ref id="B331">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2019a</year>). <article-title>Crowd behavior simulation with emotional contagion in unexpected multihazard situations</article-title>. <source>IEEE Trans. Syst. Man, Cybern. Syst.</source> <volume>51</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1109/tsmc.2019.2899047</pub-id>
</mixed-citation>
</ref>
<ref id="B332">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Deep learning for multiple object tracking: a survey</article-title>. <source>IET Comput. Vis.</source> <volume>13</volume>, <fpage>355</fpage>&#x2013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.1049/iet-cvi.2018.5598</pub-id>
</mixed-citation>
</ref>
<ref id="B333">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Duckett</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bellotto</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Online learning for human classification in 3d lidar-based tracking</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>864</fpage>&#x2013;<lpage>871</lpage>.</mixed-citation>
</ref>
<ref id="B334">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Second: sparsely embedded convolutional detection</article-title>. <source>Sensors</source> <volume>18</volume>, <fpage>3337</fpage>. <pub-id pub-id-type="doi">10.3390/s18103337</pub-id>
<pub-id pub-id-type="pmid">30301196</pub-id>
</mixed-citation>
</ref>
<ref id="B335">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Schreiberhuber</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Halmetschlager</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Duckett</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vincze</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bellotto</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Robot perception of static and dynamic objects with an autonomous floor scrubber</article-title>. <source>Intell. Serv. Robot.</source> <volume>13</volume>, <fpage>403</fpage>&#x2013;<lpage>417</lpage>. <pub-id pub-id-type="doi">10.1007/s11370-020-00324-9</pub-id>
</mixed-citation>
</ref>
<ref id="B336">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>G. S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>E. K.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>C. W.</given-names>
</name>
</person-group> (<year>2004</year>). &#x201c;<article-title>Mobile robot navigation using neural q-learning</article-title>,&#x201d;<source>Proc. 2004 Int. Conf. Mach. Learn. Cybern. (IEEE Cat. No. 04EX826)</source>, <volume>1</volume>. <fpage>48</fpage>&#x2013;<lpage>52</lpage>.</mixed-citation>
</ref>
<ref id="B337">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Urtasun</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018a</year>). &#x201c;<article-title>Pixor: real-time 3d object detection from point clouds</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>7652</fpage>&#x2013;<lpage>7660</lpage>.</mixed-citation>
</ref>
<ref id="B338">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Urtasun</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018b</year>). &#x201c;<article-title>Hdnet: exploiting hd maps for 3d object detection</article-title>,&#x201d; in <source>
<italic>Conference on robot learning</italic> (PMLR)</source>, <fpage>146</fpage>&#x2013;<lpage>155</lpage>.</mixed-citation>
</ref>
<ref id="B339">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Redmill</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>&#xd6;zg&#xfc;ner</surname>
<given-names>&#xdc;.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Top-view trajectories: a pedestrian dataset of vehicle-crowd interaction from controlled experiments and crowded campus</article-title>. <source>
<italic>IEEE Intell. Veh. Symp. (IV)</italic> (IEEE)</source>, <fpage>899</fpage>&#x2013;<lpage>904</lpage>. <pub-id pub-id-type="doi">10.1109/ivs.2019.8814092</pub-id>
</mixed-citation>
</ref>
<ref id="B340">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cadena</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Iplanner: imperative path planning</article-title>. <comment>
<italic>arXiv preprint arXiv:2302.11434</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B341">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Following social groups: socially compliant autonomous navigation in dense crowds</article-title>. <comment>
<italic>arXiv preprint arXiv:1911.12063</italic>
</comment>.</mixed-citation>
</ref>
<ref id="B342">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Roy-Chowdhury</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Sonic: safe social navigation with adaptive conformal inference and constrained reinforcement learning</article-title>. <source>arXiv Prepr. arXiv:2407.17460</source>.</mixed-citation>
</ref>
<ref id="B343">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yen</surname>
<given-names>G. G.</given-names>
</name>
<name>
<surname>Hickey</surname>
<given-names>T. W.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Reinforcement learning algorithms for robotic navigation in dynamic environments</article-title>. <source>ISA Trans.</source> <volume>43</volume>, <fpage>217</fpage>&#x2013;<lpage>230</lpage>. <pub-id pub-id-type="doi">10.1016/s0019-0578(07)60032-9</pub-id>
<pub-id pub-id-type="pmid">15098582</pub-id>
</mixed-citation>
</ref>
<ref id="B344">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Yi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Pedestrian behavior understanding and prediction with deep neural networks</article-title>,&#x201d; in <source>Computer Vision&#x2013;ECCV 2016: 14th European conference, Amsterdam, the Netherlands, October 11&#x2013;14, 2016, proceedings, part I 14</source>. <publisher-name>Springer</publisher-name>, <fpage>263</fpage>&#x2013;<lpage>279</lpage>.</mixed-citation>
</ref>
<ref id="B345">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoon</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Ym</given-names>
</name>
<name>
<surname>Jeon</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Multiple hypothesis tracking algorithm for multi-target multi-camera tracking with disjoint views</article-title>. <source>IET Image Process.</source> <volume>12</volume>, <fpage>1175</fpage>&#x2013;<lpage>1184</lpage>. <pub-id pub-id-type="doi">10.1049/iet-ipr.2017.1244</pub-id>
</mixed-citation>
</ref>
<ref id="B346">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xian</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Bdd100k: a diverse driving dataset for heterogeneous multitask learning</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>2636</fpage>&#x2013;<lpage>2645</lpage>.</mixed-citation>
</ref>
<ref id="B347">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Blukis</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Pumacay</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Krishna</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Murali</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Robopoint: a vision-language model for spatial affordance prediction for robotics</article-title>. <source>arXiv Prepr. arXiv:2406.10721</source>.</mixed-citation>
</ref>
<ref id="B348">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zakharov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kehl</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ilic</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Deceptionnet: network-driven domain randomization</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF international conference on computer vision</source>, <fpage>532</fpage>&#x2013;<lpage>541</lpage>.</mixed-citation>
</ref>
<ref id="B349">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Springenberg</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Boedecker</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Deep reinforcement learning with successor features for navigation across similar environments</article-title>,&#x201d; in <source>
<italic>2017 IEEE/RSJ international conference on intelligent robots and systems (IROS)</italic> (IEEE)</source>, <fpage>2371</fpage>&#x2013;<lpage>2378</lpage>.</mixed-citation>
</ref>
<ref id="B350">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ba&#x15f;ar</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Multi-agent reinforcement learning: a selective overview of theories and algorithms</article-title>,&#x201d; in <source>Handbook of reinforcement learning and control</source>, <fpage>321</fpage>&#x2013;<lpage>384</lpage>.</mixed-citation>
</ref>
<ref id="B351">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Bytetrack: multi-object tracking by associating every detection box</article-title>,&#x201d; in <source>European conference on computer vision</source>. <publisher-name>Springer</publisher-name>, <fpage>1</fpage>&#x2013;<lpage>21</lpage>.</mixed-citation>
</ref>
<ref id="B352">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Holloway</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Carlson</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Reinforcement learning based user-specific shared control navigation in crowds</article-title>,&#x201d; in <source>
<italic>2023 IEEE international conference on systems, man, and cybernetics (SMC)</italic> (IEEE)</source>, <fpage>4387</fpage>&#x2013;<lpage>4392</lpage>.</mixed-citation>
</ref>
<ref id="B353">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>C. W.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Se-ssd: self-ensembling single-stage object detector from point cloud</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>14494</fpage>&#x2013;<lpage>14503</lpage>.</mixed-citation>
</ref>
<ref id="B354">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tuzel</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Voxelnet: end-to-end learning for point cloud based 3d object detection</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>4490</fpage>&#x2013;<lpage>4499</lpage>.</mixed-citation>
</ref>
<ref id="B355">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Understanding collective crowd behaviors: learning a mixture model of dynamic pedestrian-agents</article-title>,&#x201d; in <source>
<italic>2012 IEEE conference on computer vision and pattern recognition</italic> (IEEE)</source>, <fpage>2871</fpage>&#x2013;<lpage>2878</lpage>.</mixed-citation>
</ref>
<ref id="B356">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Fr&#xe4;nti</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A review of motion planning algorithms for intelligent robots</article-title>. <source>J. Intelligent Manuf.</source> <volume>33</volume>, <fpage>387</fpage>&#x2013;<lpage>424</lpage>. <pub-id pub-id-type="doi">10.1007/s10845-021-01867-z</pub-id>
</mixed-citation>
</ref>
<ref id="B357">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A safe reinforcement learning approach for autonomous navigation of mobile robots in dynamic environments</article-title>. <source>CAAI Trans. Intell. Technol.</source>, <fpage>cit2.12269</fpage>. <pub-id pub-id-type="doi">10.1049/cit2.12269</pub-id>
</mixed-citation>
</ref>
<ref id="B358">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hayashibe</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A hierarchical deep reinforcement learning framework with high efficiency and generalization for fast and safe navigation</article-title>. <source>IEEE Trans. industrial Electron.</source> <volume>70</volume>, <fpage>4962</fpage>&#x2013;<lpage>4971</lpage>. <pub-id pub-id-type="doi">10.1109/tie.2022.3190850</pub-id>
</mixed-citation>
</ref>
<ref id="B359">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep reinforcement learning based mobile robot navigation: a review</article-title>. <source>Tsinghua Sci. Technol.</source> <volume>26</volume>, <fpage>674</fpage>&#x2013;<lpage>691</lpage>. <pub-id pub-id-type="doi">10.26599/tst.2021.9010012</pub-id>
</mixed-citation>
</ref>
<ref id="B360">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhe</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Collision avoidance among dense heterogeneous agents using deep reinforcement learning</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>8</volume>, <fpage>57</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2022.3222989</pub-id>
</mixed-citation>
</ref>
<ref id="B361">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Confidence-aware robust dynamical distance constrained reinforcement learning for social robot navigation</article-title>. <source>IEEE Trans. Automation Sci. Eng.</source> <volume>22</volume>, <fpage>16572</fpage>&#x2013;<lpage>16590</lpage>. <pub-id pub-id-type="doi">10.1109/tase.2025.3578326</pub-id>
</mixed-citation>
</ref>
<ref id="B362">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ziebart</surname>
<given-names>B. D.</given-names>
</name>
<name>
<surname>Maas</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Bagnell</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Dey</surname>
<given-names>A. K.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Maximum entropy inverse reinforcement learning</article-title>. <source>Aaai</source> <volume>8</volume>, <fpage>1433</fpage>&#x2013;<lpage>1438</lpage>.</mixed-citation>
</ref>
<ref id="B363">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Object detection in 20 years: a survey</article-title>. <source>Proc. IEEE</source> <volume>111</volume>, <fpage>257</fpage>&#x2013;<lpage>276</lpage>. <pub-id pub-id-type="doi">10.1109/jproc.2023.3238524</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>