<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="review-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1604472</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1604472</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multimodal perception-driven decision-making for human-robot interaction: a survey </article-title>
<alt-title alt-title-type="left-running-head">Zhao et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1604472">10.3389/frobt.2025.1604472</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Wenzheng</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3043095/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gangaraju</surname>
<given-names>Kruthika</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3105352/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yuan</surname>
<given-names>Fengpei</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3023170/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>Department of Robotics Engineering, Worcester Polytechnic Institute</institution>, <addr-line>Worcester</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/200002/overview">Bruno Lara</ext-link>, Autonomous University of the State of Morelos, Mexico</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2951654/overview">Carmela Calabrese</ext-link>, Italian Institute of Technology (IIT), Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3061278/overview">Ervin Jesus Alvarez Sanchez</ext-link>, Universidad Veracruzana, Mexico</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Fengpei Yuan, <email>fyuan3@wpi.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1604472</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zhao, Gangaraju and Yuan.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhao, Gangaraju and Yuan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Multimodal perception is essential for enabling robots to understand and interact with complex environments and human users by integrating diverse sensory data, such as vision, language, and tactile information. This capability plays a crucial role in decision-making in dynamic, complex environments. This survey provides a comprehensive review of advancements in multimodal perception and its integration with decision-making in robotics from year 2004&#x2013;2024. We systematically summarize existing multimodal perception-driven decision-making (MPDDM) frameworks, highlighting their advantages in dynamic environments and the methodologies employed in human-robot interaction (HRI). Beyond reviewing these frameworks, we analyze key challenges in multimodal perception and decision-making, focusing on technical integration and sensor noise, adaptation, domain generalization, and safety and robustness. Finally, we outline future research directions, emphasizing the need for adaptive multimodal fusion techniques, more efficient learning paradigms, and human-trusted decision-making frameworks to advance the HRI field.</p>
</abstract>
<kwd-group>
<kwd>multimodal perception</kwd>
<kwd>robot decision-making</kwd>
<kwd>human-robot interaction</kwd>
<kwd>multimodal fusion</kwd>
<kwd>robust autonomy</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Human-Robot Interaction</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The integration of robots into diverse domains such as healthcare, industrial manufacturing, transportation, and domestic environments has accelerated dramatically in recent years. Across these applications, robots serve various purposes&#x2014;from providing companionship and assistance to enabling complex collaborations with human users. Despite this diversity of contexts and functions, a fundamental requirement remains consistent: robots must interact appropriately with humans in their specific operational environments. Effective human-robot interaction (HRI) depends critically on a robot&#x2019;s ability to accurately perceive and understand human users&#x2019; status, intentions, and preferences, as well as the surrounding environment, before making appropriate decisions to achieve intended goals. This perception must then inform appropriate decision-making and action planning to achieve specific interaction goals. Consequently, the integration of multimodal perception and decision-making has emerged as a cornerstone of modern HRI research.</p>
<p>However, achieving accurate perception and robust decision-making in HRI remains a significant challenge due to the inherent complexity, dynamism, and variability of human behavior (<xref ref-type="bibr" rid="B5">Amiri et al., 2020</xref>), individual preferences, habits, capabilities (<xref ref-type="bibr" rid="B34">Ji et al., 2020</xref>), and environments (<xref ref-type="bibr" rid="B23">Diab and Demiris, 2024</xref>). Recent advancements in multimodal perception models, such as those leveraging deep learning and large-scale vision-language frameworks (<xref ref-type="bibr" rid="B42">Lu et al., 2024</xref>; <xref ref-type="bibr" rid="B49">OpenAI, 2023</xref>), coupled with increased computational power, have significantly enhanced robotic capabilities in these areas (<xref ref-type="bibr" rid="B37">Kim et al., 2024</xref>; <xref ref-type="bibr" rid="B85">Zhou et al., 2025</xref>). These developments have enabled robots to process and fuse data from multiple sensory modalities&#x2014;such as vision, speech, touch, and proprioception&#x2014;to form a more comprehensive understanding of their environment and human counterparts. Despite these advancements, the integration of multimodal perception with decision-making frameworks remains an open and actively researched problem, particularly in the context of embodied intelligence for HRI.</p>
<p>While several surveys have explored aspects of HRI, such as multimodal perception (<xref ref-type="bibr" rid="B70">Wang and Feng, 2024</xref>), human behavior modeling (<xref ref-type="bibr" rid="B55">Robinson et al., 2023</xref>; <xref ref-type="bibr" rid="B54">Reimann et al., 2024</xref>), and industrial applications (<xref ref-type="bibr" rid="B24">Duan et al., 2024</xref>; <xref ref-type="bibr" rid="B33">Jahanmahin et al., 2022</xref>; <xref ref-type="bibr" rid="B11">Bonci et al., 2021</xref>), there is a notable gap in the literature. Existing reviews often focus on specific domains, such as manufacturing (<xref ref-type="bibr" rid="B24">Duan et al., 2024</xref>; <xref ref-type="bibr" rid="B70">Wang and Feng, 2024</xref>; <xref ref-type="bibr" rid="B33">Jahanmahin et al., 2022</xref>; <xref ref-type="bibr" rid="B11">Bonci et al., 2021</xref>), or narrow aspects of HRI, such as vision (<xref ref-type="bibr" rid="B55">Robinson et al., 2023</xref>) or dialogue management (<xref ref-type="bibr" rid="B54">Reimann et al., 2024</xref>). To our knowledge, no comprehensive survey has systematically examined the interplay between multimodal perception and decision-making across diverse application domains, including healthcare, manufacturing, and transportation. This gap motivates our work.</p>
<p>In this survey, we present a comprehensive review of over 2 decades of research based on Multimodal Perception-Driven Decision-Making (MPDDM) method in embodied intelligence for HRI. Our primary objective is to analyze how these systems leverage multimodal perception to enable more efficient and accurate decision-making. Specifically, we systematically examine: (1) the sources and types of multimodal sensing data, (2) methodologies for data fusion and perception, (3) decision-making frameworks, and (4) architectures that integrate perception and decision-making. Through this analysis, we identify key challenges and limitations in current approaches and propose potential directions for future research.</p>
<p>Our contributions are threefold:<list list-type="simple">
<list-item>
<p>1. Comprehensive Cross-Domain Coverage: Unlike existing surveys that focus on specific domains, our work synthesizes HRI research across diverse application areas, including industrial manufacturing, healthcare, domestic settings, transportation, and other application areas. This cross-domain perspective provides HRI researchers with a holistic understanding of the current state of technology and methodology in multimodal perception and decision-making, potentially enabling cross-pollination of ideas between domains.</p>
</list-item>
<list-item>
<p>2. Focus on Multimodal Perception: While many reviews emphasize single modalities (e.g., vision or speech), our survey highlights the growing importance of multimodal perception in robotics and HRI. We explore how integrating multiple sensory modalities can enhance perception and decision-making.</p>
</list-item>
<list-item>
<p>3. Integration of Perception and Decision-Making: Our review not only examines multimodal perception but also discusses decision-making frameworks and their integration with perception. This dual focus offers valuable insights for researchers seeking to understand the interplay between these critical components in HRI systems.</p>
</list-item>
</list>
</p>
<p>By addressing these aspects, our survey aims to serve as a foundational resource for researchers and practitioners in the HRI community, facilitating the development of more robust and context-aware robotic systems.</p>
<p>This survey is structured as follows: <xref ref-type="sec" rid="s2">Section 2</xref> introduces the study selection process, including database searching, search strategies, and filtering criteria. <xref ref-type="sec" rid="s3">Section 3</xref> presents the survey findings, discussing the role of multimodal perception in decision-making, strategies for multimodal sensing data fusion, the MPDDM framework, and decision-making methods explored in previous research. <xref ref-type="sec" rid="s4">Section 4</xref> highlights key challenges, limitations, and potential future research directions in MPDDM within the HRI domain.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methodology</title>
<sec id="s2-1">
<title>2.1 Search and selection strategy</title>
<p>To ensure a comprehensive and systematic review, we followed the Preferred Reporting Items for Systematic Reviews and Meta-Analyses (PRISMA) guidelines (<xref ref-type="bibr" rid="B47">Moher et al., 2009</xref>). <xref ref-type="fig" rid="F1">Figure 1</xref> shows the process of identification, screening, eligibility, and inclusion in this survey. Our search strategy incorporated multiple electronic databases, including Google Scholar, SpringerLink, Web of Science, IEEE Xplore, ScienceDirect, ACM Digital Library, and Scopus, to identify relevant literature on multimodal perception-driven decision-making in human-robot interaction (HRI).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>PRISMA flow diagram of the study selection process.</p>
</caption>
<graphic xlink:href="frobt-12-1604472-g001.tif">
<alt-text content-type="machine-generated">Flowchart detailing a systematic review process. Identification phase includes sources: ACM Digital Library (43), Google Scholar (502), SpringerLink (30), ScienceDirect (30), Scopus (3), Web of Science (2), IEEE Xplore (1), totaling 641 articles. Screening phase excludes 333 articles for being non-English, duplicate, or lacking key research elements, leaving 278 eligible articles. Eligibility phase excludes 212 articles for lacking multimodal perception discussion, technical details, or case studies. Included phase results in 66 articles for literature review.</alt-text>
</graphic>
</fig>
<p>We constructed Boolean search queries based on key terms and their variations to maximize relevant results. The primary search query used in all databases was: (&#x201c;multimodal perception&#x201d; OR &#x201c;multi-modal perception&#x201d; OR &#x201c;multisensory perception&#x201d;) AND &#x201c;human-robot interaction&#x201d; AND (&#x201c;decision-making&#x201d; OR &#x201c;decision making&#x201d;). To refine our search, we applied the studies published from 2004 to 2024.</p>
<p>Using this search query, we obtained 502 hits from Google Scholar, 30 hits from SpringerLink, two hits from Web of Science, one hit from IEEE Xplore, 43 hits from ACM Digital Library, 30 hits from ScienceDirect, and three hits from Scopus. After removing duplicates, 511 articles remained for screening. Upon reviewing the article abstracts and written language, we excluded 233 articles for the following reasons: (1) non-English language, or (2) lacking key research elements for this survey (multimodal perception, human-robot interaction, and decision-making). Thus, 278 articles proceeded to the eligibility review stage. From these, we selected 66 studies that met the following inclusion criteria: (1) detailed work on integrating multimodal perception and decision-making, specifically how multimodal perception aids robots in decision-making for human-robot interaction, (2) inclusion of technical implementation details, including multimodal fusion techniques and perception-driven decision-making methodologies, and (3) concrete case studies or experimental data demonstrating practical human-robot interaction applications.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>To systematically analyze the 66 selected papers following the PRISMA guidelines, we categorized and synthesized each study based on its application domain, multimodal data types, data fusion techniques, and decision-making approaches by leveraging multimodal perception. Specifically, for each paper, we (1) provide a concise summary of its application, (2) identify the types of multimodal data utilized (e.g., vision, audio, language information), (3) classify and analyze the data fusion techniques, distinguishing between Model-Agnostic and Model-Based approaches (see <xref ref-type="sec" rid="s3-3">Section 3.3</xref> for details), and (4) examine the decision-making strategies employed (see <xref ref-type="sec" rid="s3-5">Section 3.5</xref> for further discussion), with a focus on how multimodal data contributes to improved decision-making performance. A detailed breakdown of each study is presented in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<sec id="s3-1">
<title>3.1 Application domains of MPDDM in HRI</title>
<p>Multimodal Perception-Driven Decision-Making (MPDDM) plays a crucial role in various HRI applications. By integrating multimodal perception techniques with decision-making frameworks, robots can operate in dynamic and complex environments with improved adaptability, reliability, robustness, and efficiency. Based on the reviewed literature, MPDDM applications in HRI can be categorized into four primary domains: social and assistive robotics, navigation and mobile robotics, industrial collaboration robotics, and general-purpose robotics with high-level task planning and reasoning. Furthermore, the MPDDM application domains mentioned above exhibit distinct strengths and challenges. <xref ref-type="table" rid="T1">Table 1</xref> provides a structured comparison of these domains, highlighting their key advantages as well as limitations and practical challenges, to serve as a reference for future research and application design.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of key advantages and limitations/challenges for major MPDDM application domains.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Application domain</th>
<th align="left">Key advantages</th>
<th align="left">Limitations/Practical challenges</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Social and Assistive Robotics</td>
<td align="left">Enhances user engagement, supports emotion recognition and companionship, assists in healthcare and rehabilitation</td>
<td align="left">Sensitive to user diversity (age, culture, cognitive abilities), vulnerable to environmental noise (especially audio), privacy and ethical concerns regarding sensing in healthcare</td>
</tr>
<tr>
<td align="left">Navigation and Mobile Robotics</td>
<td align="left">Improves obstacle avoidance and socially-aware navigation, compensates for missing modalities in dynamic environments</td>
<td align="left">High computational load for real-time performance, robustness in crowded/occluded scenarios, variability of human social behaviors across cultures</td>
</tr>
<tr>
<td align="left">Industrial Collaborative Robotics</td>
<td align="left">Enhances worker efficiency and safety, enables object manipulation via multimodal attribute learning</td>
<td align="left">Dynamic lighting and environmental variability, human worker behavior unpredictability, cost and complexity of multimodal sensor integration</td>
</tr>
<tr>
<td align="left">General-purpose Robotics with High-level Task Planning</td>
<td align="left">Enables robots to understand complex tasks, plan flexible actions across diverse environments, and adapt dynamically based on multimodal perception/feedback</td>
<td align="left">Generalization difficulty to unseen environments, ambiguity in interpreting user intent, computational overhead for multimodal reasoning and real-time adaptation</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3-1-1">
<title>3.1.1 Social and assistive robotics</title>
<p>Social and assistive robotics are extensively employed in social services, primarily for social interaction, emotion recognition, speech-based dialogue, assistive healthcare, and rehabilitation robotics. These systems aim to enhance user experience and engagement in HRI and provide companion, care, and/or assistance. For instance, previous work designed proactive social robots capable of responding to human emotions (<xref ref-type="bibr" rid="B1">Al-Qaderi and Rad, 2018a</xref>), situational states (<xref ref-type="bibr" rid="B68">Vauf et al., 2016</xref>), and spatial cues (<xref ref-type="bibr" rid="B13">Ch et al., 2022</xref>). Similarly (<xref ref-type="bibr" rid="B67">Tang et al., 2015</xref>), developed a companion robot based on a multimodal communication architecture for the elderly. In the field of medical assistance and rehabilitation, researchers explore the potential of MPDDM, for example (<xref ref-type="bibr" rid="B77">Yuan et al., 2024</xref>), developed a social robotic framework based on Pepper robot (<xref ref-type="bibr" rid="B50">Pandey and Gelin, 2018</xref>) for assisting persons with Alzheimer&#x2019;s dementia in executing self-care tasks, aiming to enhance their ability to complete daily routines. Additionally (<xref ref-type="bibr" rid="B53">Qin et al., 2023</xref>), designed a domestic service interactive robot system, integrating touch, speech, electromyographic gestures, visual gestures, and haptic information, explicitly aiming at individuals with declined expressive abilities.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Navigation and mobile robotics</title>
<p>Autonomous navigation and mobile robotics leverage robotic autonomy and pre-acquired environmental knowledge to facilitate human convenience. For instance, autonomous mobile robots utilizing multimodal perception for obstacle avoidance and navigation (<xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>; <xref ref-type="bibr" rid="B61">Sha, 2024</xref>; <xref ref-type="bibr" rid="B69">Wang, 2023</xref>; <xref ref-type="bibr" rid="B14">Chen et al., 2020</xref>; <xref ref-type="bibr" rid="B56">Roh, 2022</xref>; <xref ref-type="bibr" rid="B73">Xie and Dames, 2023</xref>) have been extensively studied. These studies have experimentally demonstrated that multimodal perception enhances model robustness, compensating for missing sensory modalities in dynamic environments. Meanwhile, other studies, such as (<xref ref-type="bibr" rid="B51">Panigrahi et al., 2023</xref>; <xref ref-type="bibr" rid="B64">Song D. et al., 2024</xref>; <xref ref-type="bibr" rid="B63">Siva and Zhang, 2022</xref>), focus on socially aware navigation. These works integrate vision, speech, and social signal analysis to enable robots to predict pedestrian trajectories, facilitating socially adaptive and human-friendly navigation strategies.</p>
</sec>
<sec id="s3-1-3">
<title>3.1.3 Industrial collaborative robotics</title>
<p>Industrial collaborative robotics primarily aim to enhance worker efficiency and reduce labor costs by integrating collaborative robots into manufacturing processes. This field includes classic human-robot collaborative assembly tasks (<xref ref-type="bibr" rid="B34">Ji et al., 2020</xref>; <xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>; <xref ref-type="bibr" rid="B39">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Belcamino et al., 2024</xref>) and palletizing robot (<xref ref-type="bibr" rid="B9">Baptista et al., 2024</xref>). Other research focuses on leveraging multimodal information to understand and manipulate objects. For instance (<xref ref-type="bibr" rid="B79">Zhang X. et al., 2023</xref>), and (<xref ref-type="bibr" rid="B41">Lu et al., 2023</xref>) investigate multimodal attribute learning, where robots combine visual, auditory, and haptic data to classify and recognize object properties. Once object attributes are successfully identified, the next challenge is to determine how to grasp and manipulate these objects in dynamic environments. Many researchers adopt Markov Decision Processes (MDP), such as (<xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>) and (<xref ref-type="bibr" rid="B78">Zhang et al., 2021</xref>), or reinforcement learning-based models, such as (<xref ref-type="bibr" rid="B6">Balakuntala et al., 2021</xref>), to dynamically update robotic actions based on multimodal sensory feedback. More recently, end-to-end learning models, such as (<xref ref-type="bibr" rid="B80">Zhang Z. et al., 2023</xref>), have been explored for policy generation in manipulation tasks.</p>
</sec>
<sec id="s3-1-4">
<title>3.1.4 General-purpose robotics with high-level task planning and reasoning</title>
<p>Unlike domain-specific applications, some research focuses on general task planning and decision reasoning across different HRI scenarios. Here, we examine how MPDDM can enable high-level planning beyond single-modal approaches. Traditional task-planning methods in robotics rely heavily on single-modal decision systems. However, in real-world environments, robots encounter uncertainties, dynamic human interactions, and ambiguous sensory inputs, making single-modal task planning insufficient. To address this, previous studies have integrated multimodal sensing data, such as visual, auditory, linguistic, and proprioceptive data, to enhance robotic task planning and situational reasoning. For instance (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>; <xref ref-type="bibr" rid="B80">Zhang Z. et al., 2023</xref>; <xref ref-type="bibr" rid="B45">Mei et al., 2024</xref>), and (<xref ref-type="bibr" rid="B65">Song Y. et al., 2024</xref>) leverage GPT/VLM models for semantic task parsing, enabling robots to utilize large-scale vision-language models (VLMs) for end-to-end dynamic task planning. Furthermore, these systems incorporate error correction mechanisms, which allow real-time task adjustments during execution. Beyond predefined task planning, robots operating in unstructured environments must develop situational awareness (<xref ref-type="bibr" rid="B23">Diab and Demiris, 2024</xref>), so that robots can adapt their tasks based on real-time environmental states (<xref ref-type="bibr" rid="B5">Amiri et al., 2020</xref>; <xref ref-type="bibr" rid="B81">Zhang X. et al., 2024</xref>).</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Justification of multimodal perception</title>
<sec id="s3-2-1">
<title>3.2.1 Multimodal perception</title>
<p>Multimodal perception refers to the study of methods for processing heterogeneous and interconnected data, encompassing both raw signals (e.g., speech, language, images) and abstract concepts (e.g., emotions). By integrating different modalities, humans can better perceive and interpret environmental information. Multimodal perception can be categorized into six primary types: language, vision, touch, acoustic, physiological, and mobile (<xref ref-type="bibr" rid="B40">Liang et al., 2022</xref>). Over the past decade, the rapid advancement of deep learning and embodied intelligence has significantly propelled the progress of multimodal perception-driven decision-making. In particular, the emergence of large-scale foundation models such as ChatGPT and Vision-Language-Action (VLA) frameworks has led to a new peak in multimodal development. Currently, most multimodal research focuses on vision and language integration, as researchers aim to enable robots to communicate and interpret the world similarly to humans, leveraging both linguistic reasoning and visual observation to interact with their surroundings. Such cross-modal integration enhances a robot&#x2019;s ability to comprehend complex scenarios and improves system robustness in the absence of certain sensory inputs. For example, in autonomous driving, if a robot loses radar data in a complex environment, it must still navigate safely using alternative sensory inputs (Camera, etc.) (<xref ref-type="bibr" rid="B30">Grigorescu et al., 2020</xref>). Thus, understanding the cognitive processes involved in multimodal data fusion is essential for the future of embodied artificial intelligence.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Advantages of multimodal perception</title>
<p>Single-modal perception (e.g., vision-only, speech-only, or touch-only) has played a role in early research applications. Still, it remains significantly limited in real, complex human-robot interaction scenarios (<xref ref-type="bibr" rid="B31">Huang et al., 2021</xref>). The limitations can be delineated as follows: (1) Limited Information: A single sensor provides a restricted perceptual dimension (<xref ref-type="bibr" rid="B71">Wang et al., 2024</xref>), making capturing global or deep semantic information challenging. (2) Poor Robustness: Single-modal systems are highly susceptible to noise, lighting changes, occlusions, or hardware failures, leading to performance degradation (<xref ref-type="bibr" rid="B71">Wang et al., 2024</xref>). (3) Lack of Accuracy and Generalization: In complex, dynamic environments, single-modal algorithms struggle to maintain high accuracy or adapt quickly (<xref ref-type="bibr" rid="B31">Huang et al., 2021</xref>). (4) Inability to Capture Multifaceted Human/Environment Information: Human language, emotions, intentions, and actions involve multiple signals, which a single modality alone cannot fully comprehend (<xref ref-type="bibr" rid="B66">Su et al., 2023</xref>).</p>
<p>Due to these limitations, researchers have increasingly focused on multimodal perception (<xref ref-type="bibr" rid="B25">Duncan et al., 2024</xref>) in recent years, aiming to integrate information from different types of sensors to handle complex, dynamic HRI scenarios. By integrating different modalities such as text/speech, vision, audio, touch, and physiological signals, multimodal perception offers advantages such as information complementarity and enhanced robustness. For instance (<xref ref-type="bibr" rid="B1">Al-Qaderi and Rad, 2018a</xref>), and (<xref ref-type="bibr" rid="B16">Churamani et al., 2020</xref>) demonstrated that combining auditory cues with visual inputs improved the accuracy of recognizing personal emotion and location compared to under visual-based detection conditions. Similarly (<xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>), leveraged vision and speech to enhance the accuracy and robustness of human detection and interaction in complex scenarios. Furthermore (<xref ref-type="bibr" rid="B69">Wang, 2023</xref>), and (<xref ref-type="bibr" rid="B36">Khandelwal et al., 2017</xref>) found that multisensory data from both RGB-D cameras and LiDAR sensors mitigated the instability of visual-only systems and improved the robustness of navigation. Beyond robustness, multimodal perception also enhances contextual and semantic comprehension in HRI scenarios. Just like humans, who rely on the integration of multiple modalities (e.g., hearing, vision, smell) to better interpret their surroundings, multimodal perception enables robots to achieve a more comprehensive, accurate understanding of environmental states. For example (<xref ref-type="bibr" rid="B78">Zhang et al., 2021</xref>), enabled robots to explore and describe objects in the environment as humans by using audio, haptics, and vision, this approach improved object description accuracy by 50% compared to vision-only exploration.</p>
<p>In summary, multimodal perception not only addresses the inherent weaknesses of unimodal perception but also broadens the scope of MPDDM applications, paving the way for richer human-robot collaboration. However, alongside these advantages, multimodal information also introduces challenges, such as the complexity of data fusion, multimodal representation learning&#x2014;how to utilize multimodal information effectively, alignment&#x2014;how to model connections across modalities to ensure accurate understanding and integration, and reasoning&#x2014;how different modalities interact to influence the decision-making process. These challenges and methods will be explored in <xref ref-type="sec" rid="s3-3">Sections 3.3</xref>,<xref ref-type="sec" rid="s3-4">3.4</xref>.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Multimodal sensing data fusion strategies</title>
<p>Multimodal data fusion represents the cornerstone of effective perception-driven decision-making in human-robot interaction systems. This process involves systematically combining information streams from diverse sensing modalities to have a comprehensive representation of the environment. Successful fusion strategies enable robots to overcome the limitations of individual sensors, enhance perceptual robustness in challenging conditions, and develop a more complete understanding of complex human-robot interaction scenarios. In this section, we examine the primary approaches to multimodal data fusion following the framework established by (<xref ref-type="bibr" rid="B7">Baltru&#x161;aitis et al., 2018</xref>). The fusion methodologies can be broadly categorized into two fundamental classes: model-agnostic methods and model-based methods. Model-agnostic approaches offer flexibility across different learning paradigms, while model-based techniques integrate fusion mechanisms directly within the learning architecture.</p>
<p>Model-agnostic methods typically operate at distinct stages of the perception pipeline, with fusion occurring at the data level (early fusion), feature level (intermediate fusion), decision level (late fusion), or through hybrid combinations spanning multiple processing stages. Meanwhile, model-based methods leverage the inherent capabilities of neural networks, kernel methods, or probabilistic graphical models to learn optimal fusion strategies during the training process. The following subsections detail these approaches, examining their theoretical foundations, implementation considerations, and relative advantages in various HRI contexts.</p>
<sec id="s3-3-1">
<title>3.3.1 Model-agnostic methods</title>
<p>Early Fusion (Data-Level): In early fusion, raw or minimally processed data from different modalities are combined into a single input representation at the earliest stage. For example, in a long-term social interaction bartending robot task (<xref ref-type="bibr" rid="B57">Rossi et al., 2024</xref>), enhanced the robot&#x2019;s natural interactive operations by incorporating speech and facial expressions. Similarly (<xref ref-type="bibr" rid="B48">Nan et al., 2019</xref>), employed early fusion of RGB and depth images by aligning them based on time frames to improve elderly action recognition. The early-fusion method allows the subsequent model (or pipeline) to learn cross-modal correlations directly from the original data, but it may become challenging to handle large discrepancies or noise across modalities.</p>
<p>Intermediate Fusion (Feature-Level): Features are first extracted independently from each modality, and then these feature representations are fused. This approach balances flexibility and complexity&#x2014;each modality can be processed separately with tailored feature extraction techniques, and the combined feature space typically captures richer, modality-specific information before final decision-making (<xref ref-type="bibr" rid="B34">Ji et al., 2020</xref>). Demonstrated that feature-level fusion of vision, depth, and inertial sensors enables reliable perception by capturing information about humans, robots, and the environment in industrial human-robot collaboration (HRC) (<xref ref-type="bibr" rid="B58">Schmidt-Rohr et al., 2008a</xref>). Improved the accuracy of &#x201c;person of interest&#x201d; recognition and ensured stable autonomous navigation by converting raw sensory inputs (speech, human activity from RGB-D, and LiDAR) into probability distributions, enhancing dynamic confidence from different feature levels (<xref ref-type="bibr" rid="B60">Scicluna et al., 2024</xref>). Showed that aligning 2D feature bounding boxes from RGB with LiDAR depth features prevented false positive detections from leading to incorrect decisions. Similarly, studies such as (<xref ref-type="bibr" rid="B8">Banerjee et al., 2018</xref>), (<xref ref-type="bibr" rid="B83">Zhao et al., 2024</xref>), and (<xref ref-type="bibr" rid="B22">Deng et al., 2024</xref>) enhanced perception capabilities by employing various fusion strategies, including weighting, concatenation, and heuristic-based algorithms.</p>
<p>Late Fusion (Decision-Level): Late fusion focuses on merging the outputs (i.e., decisions or predictions) from multiple models or classifiers. Each modality is modeled separately and the final results are combined (e.g., by voting, averaging, or a learned ensemble). For example (<xref ref-type="bibr" rid="B62">Siqueira et al., 2018</xref>), separately predicted emotion and language models, then evaluated the recognition results using a decision framework to resolve emotional mismatches. Similarly (<xref ref-type="bibr" rid="B29">Granata et al., 2013</xref>), merged information extracted from four detectors using weighted criteria based on the field of view, reducing the instability of motion prediction when sensor data was incomplete. Late fusion often offers greater robustness if one modality performs poorly, but it may miss certain cross-modal interactions that arise earlier in the data or feature space.</p>
<p>Hybrid Fusion (Combining Multiple Stages): Hybrid fusion integrates multiple fusion strategies. For instance, combining early or intermediate fusion with late fusion. By doing so, it aims to leverage the best of both worlds: capturing cross-modal correlations and ensuring robust final decisions. For example (<xref ref-type="bibr" rid="B61">Sha, 2024</xref>), performed feature-level fusion by assigning weights to RGB object detection, depth cameras, and ultrasonic sensors, then integrated obstacle detection and path planning module outputs at the decision level. Similarly (<xref ref-type="bibr" rid="B21">Dean-Leon et al., 2017</xref>), applied fusion at multiple stages, including signal level, feature level, and symbolic representation, to enhance collision avoidance, compliance, and grasping strategies. Overall, hybrid fusion approaches aim to leverage the advantages of different fusion stages and integrate them effectively.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Model-based fusion</title>
<p>In this category, the fusion process is driven by a learned model&#x2014;often nonlinear&#x2014;such as probabilistic methods, kernel-based methods, neural networks (e.g., CNNs, Transformers), or graph-based models. These approaches learn how to integrate or attend to relevant information across modalities through training, allowing more adaptive and potentially more powerful multimodal representations. For example (<xref ref-type="bibr" rid="B19">Da&#x11f;larl&#x131;, 2020</xref>), utilized probabilistic reasoning and attention mechanisms to integrate multimodal perception data (<xref ref-type="bibr" rid="B78">Zhang et al., 2021</xref>). Dynamically constructed a partially observable Markov decision process (POMDP) that integrates information from different sensory modalities and actions to compute the optimal policy.</p>
<p>In the domain of neural network approaches (<xref ref-type="bibr" rid="B76">Yu, 2021</xref>), employed CNNs and GANs for gesture/facial synthesis and a hybrid classifier for emotion recognition (<xref ref-type="bibr" rid="B69">Wang, 2023</xref>). Utilized an attention mechanism to integrate visual and temporal multimodal features (<xref ref-type="bibr" rid="B74">Yas et al., 2024</xref>). extracted RGB and depth data and fused them with skeletal features using a context attention mechanism. Similarly (<xref ref-type="bibr" rid="B2">Al-Qaderi and Rad, 2018b</xref>), used spiking neural networks (SNN) to process feature vectors from different modalities, including RGB (FERET), RGB-D (TIDIGITS), and RGB-D (3D body and depth).</p>
<p>Other learning-based approaches include reinforcement learning and graph-based methods (<xref ref-type="bibr" rid="B18">Cuay&#xe1;huitl, 2020</xref>). Applied deep reinforcement learning (DQN) to fuse visual perception and speech interaction (<xref ref-type="bibr" rid="B26">Ferreira et al., 2012</xref>). Employed a graph-based Bayesian hierarchical model to fuse visual and auditory perception. Similarly (<xref ref-type="bibr" rid="B32">Ivaldi et al., 2013</xref>), integrated vision, audition, and proprioception using graph-based incremental learning and sensorimotor loops.</p>
<p>More recently, large language models (LLMs) have been leveraged for multimodal fusion (<xref ref-type="bibr" rid="B46">Menezes, 2024</xref>). Leveraged, Generative Image-to-text Transformer (GIT) and GPT-4 for cross-modal alignment of visual, textual, and auditory inputs (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>). Utilized GPT-4V for feature extraction and decision-making based on visual and contextual inputs. And (<xref ref-type="bibr" rid="B43">Ly et al., 2024</xref>) employed an LLM-based planner to generate action sequences by integrating recognition results with motion feasibility.</p>
<p>In summary, multimodal data fusion can be achieved through various strategies&#x2014;from a simple early fusion of raw data to advanced hybrid and model-based methods that dynamically learn cross-modal interactions. <xref ref-type="table" rid="T2">Table 2</xref> summarizes the main fusion strategies discussed above, along with their key advantages and limitations, to provide a concise reference for readers. These approaches provide the foundation for robust and context-rich perception in HRI. In the next section, we explore how these fused representations are integrated into decision-making architectures, enabling robots to leverage the full potential of multimodal inputs for more intelligent and adaptive behavior.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary of multimodal fusion strategies, their advantages, and limitations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Fusion strategy</th>
<th align="left">Advantages</th>
<th align="left">Limitations</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Early Fusion (Data-Level)</td>
<td align="left">Enables learning of cross-modal correlations from original data</td>
<td align="left">Sensitive to modality noise, requires careful alignment</td>
</tr>
<tr>
<td align="left">Intermediate Fusion (Feature-Level)</td>
<td align="left">Balances flexibility and richer modality-specific information</td>
<td align="left">Dependent on effective feature extraction for all modalities</td>
</tr>
<tr>
<td align="left">Late Fusion (Decision-Level)</td>
<td align="left">Robust to failure in individual modalities</td>
<td align="left">May miss useful cross-modal interactions present at earlier stages</td>
</tr>
<tr>
<td align="left">Hybrid Fusion</td>
<td align="left">Leverages strengths of different fusion stages</td>
<td align="left">More complex implementation and design effort</td>
</tr>
<tr>
<td align="left">Model-Based Fusion</td>
<td align="left">Adaptable and powerful multimodal representations</td>
<td align="left">Typically requires large datasets and can exhibit lower interpretability compared to simpler fusion strategies</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Integration architectures for multimodal perception and decision-making</title>
<p>Multimodal perception can be integrated into decision-making processes for HRI through various architectural frameworks, ranging from conventional linear pipelines to more advanced adaptive models incorporating feedback loops and end-to-end learning. The selection of an appropriate architecture is contingent on multiple factors, including real-time processing constraints, system complexity, and the degree of adaptability required for a given task. For instance, simple feedforward pipelines may be suitable for low-latency applications (<xref ref-type="bibr" rid="B68">Vauf et al., 2016</xref>). In contrast, end-to-end frameworks or feedback architecture are often preferred in dynamic and uncertain environments where continuous adaptation is necessary (<xref ref-type="bibr" rid="B45">Mei et al., 2024</xref>; <xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>). Therefore, in this section, we analyze the rationale behind the selection of each architectural approach by synthesizing insights from selected papers and empirical findings. We will discuss how different architectures align with specific HRI tasks, the trade-offs they present in performance and adaptability. In multimodal perception-driven decision systems, both academia and industry commonly use five types of high-level architectures to integrate information and execute action decisions for robotics: pipeline architecture, feedback-loop architecture, modular architecture, end-to-end architecture, and hybrid architecture. The first four basic types of architecture have been shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, which illustrates the workflow of each approach.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Four basic integration architectures for multimodal perception and decision-making in human-robot interaction. <bold>(a)</bold> Pipeline architecture. <bold>(b)</bold> Feedback architecture. <bold>(c)</bold> Modular architecture. <bold>(d)</bold> End to end architecture.</p>
</caption>
<graphic xlink:href="frobt-12-1604472-g002.tif">
<alt-text content-type="machine-generated">Diagram illustrating four architecture approaches for multimodal input processing. (a) Pipeline Architecture: Inputs are fused, leading to decision making. (b) Feedback Architecture: Feedback loop for fusion adjustment. (c) Modular Architecture: Separate modules for perception, planning, and control integrating through a hub. (d) End-to-End Architecture: Multimodal inputs process directly in a model to produce decision output. Each approach routes inputs like images, speech, and gestures through different pathways for decision making.</alt-text>
</graphic>
</fig>
<sec id="s3-4-1">
<title>3.4.1 Pipeline architecture</title>
<p>The pipeline processing architecture allows multiple modalities to be processed simultaneously, reducing processing latency. By handling different sensory inputs in parallel and integrating them through a coordination layer, the system can output multimodal results in real-time, feeding them directly into the planning and decision-making module, as illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref> (see Subfigure a). This architecture is particularly advantageous in real-time interactive scenarios, where synchronized multimodal processing ensures fast and adaptive responses (<xref ref-type="bibr" rid="B68">Vauf et al., 2016</xref>). Implemented a multi-channel parallel processing architecture to detect whether a person intends to initiate interaction with the robot, which enables social companion robots to respond to human behavior more naturally in real-time. They enabled the same robot to acquire and process multiple sensory inputs in parallel, integrating data from different modalities: Laser scanner: 270&#xb0; field of view, updated every 80 m (12.5Hz), mounted at the base of the Kompa&#xef; robot (captures spatial position and distance); Kinect sensor: RGB video (30Hz) for skeleton tracking and facial detection; Depth images (30Hz) for enhanced skeleton tracking; Microphone array: 8Hz for sound source localization and voice activity detection. All features were temporally aligned using an 80 m (12.5Hz) baseline, with data from different modalities fused via temporal synchronization and feature concatenation. The unified multimodal representation was then fed into a classifier (e.g., SVM or neural networks) to recognize interaction intent. They claimed this approach allows the robot to robustly and efficiently infer user engagement, ensuring real-time, natural, and adaptive responses in HRI scenarios.</p>
</sec>
<sec id="s3-4-2">
<title>3.4.2 Feedback-loop architecture</title>
<p>As illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref> (see Subfigure b), this architecture incorporates feedback mechanisms where outputs from later stages (e.g., decision-making) influence earlier stages (e.g., perception or sensing). This approach enables adaptive and context-aware behavior, improving robustness by allowing the system to refine its perception based on decision outcomes. Such an architecture is particularly well-suited for real-time multimodal perception scenarios in complex human-robot interaction. For example, social robots continuously perceive user emotions and dynamically adjust their dialogue strategies (<xref ref-type="bibr" rid="B13">Ch et al., 2022</xref>), or collaborative robots refine grasping operations in real-time using force and visual feedback (<xref ref-type="bibr" rid="B34">Ji et al., 2020</xref>; <xref ref-type="bibr" rid="B79">Zhang X. et al., 2023</xref>) introduces a pipeline for robotic interaction and perception&#x2014;the Multimodal Embodied Attribute Learning (MEAL) framework. MEAL enables robots to perceive object attributes&#x2014;such as color, weight, and empty&#x2014;through sequential multimodal exploratory behaviors (e.g., observing, lifting, and shaking). The framework is built on a Partially Observable Markov Decision Process (POMDP) for object attribute recognition, structured as follows: Action Selection: The robot selects the next exploratory action (e.g., look, grasp, shake) based on the current environmental state and its belief about the object&#x2019;s attributes. Information Acquisition: The robot executes the chosen action and collects new sensory data across multiple modalities (e.g., visual, auditory, and tactile features). Belief Update: The system integrates new observations and user feedback (e.g., confirming or correcting attribute recognition) to update the POMDP belief state. In ONLINE-MEAL scenarios, newly collected data (features &#x2b; labels) are also added to the training set to improve future perception models. Decision Re-evaluation: After updating its belief or model, the system reassesses whether further exploration is needed or if it can confidently report results, forming a closed-loop perception-decision-feedback cycle. Similarly (<xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>), employs multi-source perception (RGB-D, QR codes, wheel odometry, etc.) to obtain the robot&#x2019;s current state and detect obstacles ahead. The system continuously feeds obstacle location and distance information to the &#x201c;Safe Manipulation&#x201d; module in real time. Based on the relative distance and direction between the obstacle and the robot, this module dynamically adjusts the robot&#x2019;s speed or triggers braking to ensure safe operation.</p>
</sec>
<sec id="s3-4-3">
<title>3.4.3 Modular architecture</title>
<p>The system is divided into independent, self-contained modules, each responsible for a specific function (e.g., sensing, perception, decision-making). Therefore, This architecture is well-suited for scalable robotic systems, particularly industrial collaborative robots. It consists of independent modules, such as a vision detection module (for object localization), a force/tactile sensing module (to ensure safe interaction with humans or objects), a motion planning module (for generating robotic arm trajectories), and a high-level decision-making module (for task allocation and anomaly handling). Each module can be individually upgraded or replaced without affecting the overall system framework. A typical implementation involves clear interfaces or communication protocols between modules, such as topics or services in ROS (Robot Operating System). For instance (<xref ref-type="bibr" rid="B36">Khandelwal et al., 2017</xref>), developed a modular and hierarchical general-purpose platform that integrates various independent modules, including mapping, robot actions, task planning, navigation, perception, and multi-robot coordination. Each module has a well-defined function and can be replaced without affecting the overall system. For example, the perception module uses a Kinect camera for human and object detection and LiDAR for environment perception. During real-world operation, the robot continuously perceives its surroundings, updating its knowledge state (via knowledge representation and reasoning nodes) and executing actions based on high-level planning. The key advantages of this architecture include ease of maintenance, upgradability, and flexibility for expanding to more complex tasks. However, a potential drawback is the added system overhead due to module coordination.</p>
</sec>
<sec id="s3-4-4">
<title>3.4.4 End-to-end architecture</title>
<p>End-to-end methods typically involve designing a unified neural network that directly maps sensor inputs to decision-making or action, without requiring manually engineered intermediate steps (<xref ref-type="bibr" rid="B43">Ly et al., 2024</xref>). Introduced an &#x201c;end-to-end&#x201d; high-level task planning architecture, which processes natural language instructions from humans and integrates visual perception and action feasibility verification. The system combines user input (U), visual observations (O), and feasibility scores (F) into a multimodal context, which is then fed into a fine-tuned Mistral 7B model to automatically generate and execute robot skill sequences (e.g., pick object, move to location, place object), directly mapping to atomic operations from the robot&#x2019;s existing skill library. Additionally, the framework incorporates failure recovery and a human-in-the-loop mechanism. If the visual perception or feasibility detection module fails, the LLM prompts the user for guidance on handling the failure. The user can then provide new descriptions, suggest alternative objects, or manually reposition objects. The LLM subsequently generates a revised action sequence to ensure task completion. The advantage of this architecture is that it eliminates the need for complex feature engineering and pipeline framework construction. However, its drawbacks include lower interpretability and higher requirements for hardware and algorithms.</p>
</sec>
<sec id="s3-4-5">
<title>3.4.5 Hybrid architecture</title>
<p>The hybrid architecture combines elements from multiple architectural paradigms to leverage their respective strengths. For example, it may incorporate aspects of parallel processing, pipeline structures, and feedback loops. Effectively integrating these mechanisms at different stages or levels can balance real-time performance and flexibility. One example is the brain-inspired multimodal perception system proposed by (<xref ref-type="bibr" rid="B1">Al-Qaderi and Rad, 2018a</xref>), which follows a hybrid architecture approach: The system processes different modalities&#x2014;vision (RGB and depth cameras) and audio (microphones)&#x2014;in parallel through Dedicated Processing Units (DPUs), each responsible for specialized feature extraction and classification. Within each modality, information flows through a fixed-sequence pipeline, such as facial detection, skeletal tracking, feature extraction, and classification. Moreover, at a higher level, the system integrates spiking neural networks (SNNs) for temporal binding and top-down influences (e.g., using QR codes to infer possible identities), enabling a feedback loop to refine lower-level processing dynamically. This approach allows the system to handle multi-source inputs in parallel, while selectively activating or constraining lower layers based on intermediate recognition results. By doing so, it reduces unnecessary computation and improves response speed. Such a hybrid framework has demonstrated high adaptability and flexibility in dynamic HRI scenarios.</p>
</sec>
</sec>
<sec id="s3-5">
<title>3.5 Decision-making methodologies</title>
<p>In the previous subsection, we examined how multimodal perception provides rich context for decision-making by integrating vision, audio, tactile, and other sensory inputs. However, perception alone does not complete the cycle: once the environment is understood, the robot must decide how to act in response (see <xref ref-type="sec" rid="s3-4">Section 3.4</xref> for Integration Architectures). Decision-making is a fundamental capability in intelligent systems, enabling robots and AI agents to infer contextual information by perceiving the environment, and then generate appropriate actions. For MPDDM, the system integrates information from multiple sensory modalities&#x2014;such as vision, audio, touch, and language&#x2014;to enhance robustness and adaptability in dynamic environments. With the rapid advancement of deep learning, symbolic reasoning, and probabilistic models, decision-making methods have become more adaptive and learning-based paradigms. <xref ref-type="table" rid="T3">Table 3</xref> summarizes the decision-making method used by the MPDDM system. However, selecting the appropriate decision-making framework depends on the goal of the task, the level of uncertainty in the environment, and the training data. In this subsection, we summarize the decision-making methodologies that have been studied in HRI, from seven decision-making perspectives. Under each perspective, we presented each method&#x2019;s distinct strengths and trade-offs in handling environmental uncertainty, data requirements, and the complexity of collaboration:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Learning-Based Paradigm: Supervised Learning, Reinforcement Learning, Imitation Learning, etc.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Problem Formulation Based: Markov Decision Process (MDP), Partially Observable MDP (POMDP), Game Theory, etc.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Symbolic/Logic-Based Approaches: Rule-Based Systems, Automated Planning (STRIPS, HTN), Knowledge Representation, etc.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Probabilistic Methods: Bayesian Networks, Hidden Markov Models (HMMs), etc.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Search-Based Planning: Path Planning, Monte Carlo Tree Search (MCTS), etc.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Generative AI Decision-Making: LLM/VLM-based Decision-Making, Generative Adversarial Imitation Learning (GAIL), etc.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Hybrid Approaches: Combine multiple methodologies</p>
</list-item>
</list>
</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Summary of multimodal perception and decision-making methods in HRI.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">References</th>
<th align="center">Application</th>
<th align="center">Multimodal data</th>
<th align="center">Fusion technique</th>
<th align="center">Decision making process</th>
<th align="center">Decision making method</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">
<xref ref-type="bibr" rid="B59">Schmidt-Rohr et al. (2008b)</xref>
</td>
<td align="center">Reasoning system for a multi-modal service robot with human-robot interaction</td>
<td align="center">RGB-D; Audio; Self-SLAM; Force &#x2b; Torque</td>
<td align="center">Bayesian Filtering; Feature Filter; POMDP</td>
<td align="center">POMDP-based decision-making, utilizing fused multimodal belief states for optimal task execution</td>
<td align="center">POMDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B12">Cai et al. (2024)</xref>
</td>
<td align="center">Autonomous navigation, intelligent control, and social interaction</td>
<td align="center">RGB-D; Time-Series Data; Audio; Textual Data</td>
<td align="center">Multimodal Data Integration; Attention Mechanisms</td>
<td align="center">Prediction Models (CNN &#x2b; LSTM &#x2b; RL)</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B51">Panigrahi et al. (2023)</xref>
</td>
<td align="center">Multimodal perception-driven decision-making in social robot navigation</td>
<td align="center">RGB; LiDAR</td>
<td align="center">Outputs of the RNNs from the point cloud and image modules are concatenated and passed through the fusion process</td>
<td align="center">Uses an MLP to generate these waypoints based on multimodal fused features; Predicts linear and angular velocities for immediate motion</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B19">Da&#x11f;larl&#x131; (2020)</xref>
</td>
<td align="center">Social robots that cognitive perception to interact with humans and navigate dynamic environments as personal assistants</td>
<td align="center">Visual; Auditory; Tactile; Laser Range</td>
<td align="center">Probabilistic reasoning and attention mechanisms to integrate multimodal perception data; Linking semantic relations and perception data at the feature level through knowledge anchoring</td>
<td align="center">Dynamic world model integrates multimodal data to facilitate joint attention and cognitive perception to support decision-making (HRI) in social robots</td>
<td align="center">Bayesian Networks</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B46">Menezes (2024)</xref>
</td>
<td align="center">Collaborative target detection involving human operators, virtual robots, and real robots in a mixed virtual-real environment</td>
<td align="center">RGB-D; Auditory; Textual</td>
<td align="center">Use of GIT and GPT-4 for cross-modal alignment of visual and textual/auditory inputs</td>
<td align="center">MuModaR framework integrates multimodal inputs to enable real-time feedback-driven decision-making</td>
<td align="center">LLM-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B79">Zhang et al. (2023a)</xref>
</td>
<td align="center">Object-centric learning for robots interacting directly with physical objects to determine their properties</td>
<td align="center">vision, audio, haptics</td>
<td align="center">Hybrid Fusion; Combines multimodal sensory inputs from behaviors into a feature set for attribute identification; Support Vector Machines (SVMs) and weighted feature combinations for modality-specific attribute classification; Observations are aggregated and updated in belief states through probabilistic reasoning (MOMDPs)</td>
<td align="center">MOMDPs handle mixed observability by separating fully observable components (e.g., object grasp success) and partially observable components (e.g., color)</td>
<td align="center">MOMDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B82">Zhang et al. (2024b)</xref>
</td>
<td align="center">Navigation system applied in healthcare, logistics, and domestic settings, enhancing the interaction capabilities of mobile robots</td>
<td align="center">RGB-D; QR sensors; Wheel Encoder</td>
<td align="center">Transformers process RGB-D patches for obstacle identification and segmentation; CNNs aggregate multi-layer features for depth estimation; Outputs from QR, RGB-D, and wheel encoders are fused for state estimation and decision-making</td>
<td align="center">Integrates multimodal data for real-time navigation and safe manipulation in complex environments, adapting robot speed and behavior based on perceived obstacles</td>
<td align="center">Rule-based system</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B1">Al-Qaderi and Rad (2018a)</xref>
</td>
<td align="center">Human-Robot Interaction: Implemented in social robots for person recognition, emotion recognition, and context-dependent response</td>
<td align="center">RGB-D (Kinect); RGB camera (USB3 Flea3); microphone (Rode VideoMic); laser rangefinder (SICK LMS-200); sonar sensors</td>
<td align="center">Model-Based Fusion: Liquid State Machines (LSM) and Leaky Integrate-and-Fire (LIF) neurons to perform feature integration</td>
<td align="center">Person recognition using facial, body, and voice features. Adaptive multimodule recognition where missing modalities (e.g., face occlusion) are compensated by available features (e.g., voice)</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B68">Vauf et al. (2016)</xref>
</td>
<td align="center">Determining interaction initiation for companion robots based on multimodal perception</td>
<td align="center">RGB-D (Kinect); Microphone (Kinect), Laser Range Finder</td>
<td align="center">Artificial Neural Networks (ANNs) and Support Vector Machines (SVMs) for fusing multimodal features (spatial positions, body postures, and acoustic data) and classifying engagement states</td>
<td align="center">Enhancing decision-making robustness in noisy environments by accurately predicting user interaction initiation intent through integrating multimodal data such as body posture, spatial cues, and audio signals</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B28">Granata et al. (2012)</xref>
</td>
<td align="center">Human detection and interaction systems in complex environments</td>
<td align="center">RGB-D (Kinect), Infrared (IR), Laser, Microphone</td>
<td align="center">Independent feature extraction by different detectors (e.g., leg detection via LiDAR, RGB &#x2b; IR for upper body), with spatial alignment and weighting applied to the features</td>
<td align="center">Confidence values from multimodal fusion guide decisions, ensuring robust detection even under partial occlusion or noise</td>
<td align="center">Rule-based system</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B67">Tang et al. (2015)</xref>
</td>
<td align="center">Human-friendly robot partner to assist the elderly</td>
<td align="center">RGB-D (Kinect); SunSPOT (room brightness, temperature, and human activity logs); Microphone</td>
<td align="center">Spiking neural networks (SNNs) and growing neural gas (GNG) algorithms to process and fuse gesture recognition data</td>
<td align="center">Robot decision-making (e.g., selecting dialogue topics or responding to gestures) is guided by an emotion model computed from multimodal inputs, enabling context-appropriate natural interactions</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B61">Sha (2024)</xref>
</td>
<td align="center">A novel multimodal perception navigation system for a real open environment</td>
<td align="center">RGB-D (Intel RealSense D456); RGB-IR; 5x Ultrasonic Sensors; GPS; IMU; Audio Data</td>
<td align="center">Features extracted from depth cameras, ultrasonic sensors, and object detection algorithms are weighted and spatially aligned. Outputs from obstacle detection and path planning modules are fused at the decision level based on predefined rules</td>
<td align="center">Achieving comprehensive environmental understanding through multimodal perception with dynamic updates to navigation and interaction strategies</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B4">Amiri et al. (2018)</xref>
</td>
<td align="center">Planning for robotic arms using MOMDP with multimodal perception</td>
<td align="center">Visual Data; Haptic Data; Proprioceptive Data; Audio Data; (Thomason16, Sinapov14 Dataset)</td>
<td align="center">Mixed Observability Markov Decision Processes (MOMDPs) are used to integrate multimodal data streams</td>
<td align="center">Modeling fully and partially observable state variables with MOMDP updated through multimodal sensory feedback to optimize subsequent decisions</td>
<td align="center">MOMDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B13">Ch et al. (2022)</xref>
</td>
<td align="center">An emotion-driven social robot</td>
<td align="center">RGB (Facial expressions); Speech Data (speech of the users); (Aff-Wild, AFEW-VA, RAVDESS, SAVEE datasets)</td>
<td align="center">MCCNN (Multi-Channel Convolutional Neural Network) extracts and classifies multimodal feature representations</td>
<td align="center">Multimodal perception integrates visual and vocal emotional information, combined with emotional memory and core affect, to enhance negotiation adaptability and user experience</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B34">Ji et al. (2020)</xref>
</td>
<td align="center">A human-robot collaboration framework using multimodal inputs based on POMDP planning</td>
<td align="center">RGB-D camera; IMUs; Force Sensors</td>
<td align="center">Features from image-based and non-image-based sensors are extracted and combined into unified representations for human behavior recognition and motion planning</td>
<td align="center">The POMDP framework uses multimodal data to update beliefs about the human&#x2019;s state and intentions, enabling adaptive motion planning</td>
<td align="center">POMDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B27">Forlini et al. (2024)</xref>
</td>
<td align="center">A robot-assisted assembly planner based on Large Multimodal Models (LMM)</td>
<td align="center">RGB (2x Intel RealSense D435i); Language (all assembly components, their precedence relationships)</td>
<td align="center">GPT-4V for feature extraction and decision-making based on visual and contextual inputs</td>
<td align="center">Visual and context-aware enables accurate identification of components and their assembly states</td>
<td align="center">VLM-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B76">Yu (2021)</xref>
</td>
<td align="center">A social robot for natural human-robot interaction</td>
<td align="center">RGB-D; Thermal cameras; Audio Data</td>
<td align="center">CNN, GANs for gesture/facial synthesis; hybrid classifier for emotion recognition</td>
<td align="center">Enables natural, emotion-sensitive HRI through synchronized multimodal behavior generation</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B69">Wang (2023)</xref>
</td>
<td align="center">Robust and privacy-preserving autonomous driving</td>
<td align="center">RGB; LiDAR; Radar</td>
<td align="center">ResNet-50 extracts high-level visual features, the LSTM captures temporal dependencies, attention mechanism integrates these multimodal features</td>
<td align="center">Multimodal sensing reduced ambiguity in perception (combining complementary data like RGB and LiDAR) and robustness against single-sensor failures</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B36">Khandelwal et al. (2017)</xref>
</td>
<td align="center">A robot for human-robot interaction in open environments</td>
<td align="center">RGB (Kinect); LIDAR (Velodyne); Language</td>
<td align="center">Visual features, spatial maps, and language representations are processed independently and then fused at the decision-making level</td>
<td align="center">Combines probabilistic reasoning and planning (CORPP) to infer missing or ambiguous information, reducing errors in understanding and navigation</td>
<td align="center">Path Planning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B5">Amiri et al. (2020)</xref>
</td>
<td align="center">A sequential decision-making framework for robots</td>
<td align="center">RGB-D; contextual knowledge</td>
<td align="center">The classifier&#x2019;s output (probabilistic state estimation) is combined with declarative contextual knowledge using P-log (logical-probabilistic reasoning)</td>
<td align="center">The POMDP planner uses state estimates from sensor data and contextual knowledge as priors to determine the optimal actions for proactive human-robot interaction</td>
<td align="center">POMDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B78">Zhang et al. (2021)</xref>
</td>
<td align="center">A multimodal exploratory action interaction robotic arm</td>
<td align="center">RGB; Haptics; Audio</td>
<td align="center">The dynamically constructed POMDP integrates information from different sensory modalities and actions to compute the policy</td>
<td align="center">The decision-making is driven by the POMDP, which integrates multimodal sensory data to refine the robot&#x2019;s belief state and compute the best action policy</td>
<td align="center">POMDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B74">Yas et al. (2024)</xref>
</td>
<td align="center">A multi-agent motion prediction framework</td>
<td align="center">RGB-D; Skeleton Data</td>
<td align="center">Extracts RGB, depth and fuses them with skeletal features using a context attention mechanism</td>
<td align="center">Predicting future trajectories and joint positions of all agents in the scene</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B23">Diab and Demiris (2024)</xref>
</td>
<td align="center">An assistive HRI robot for daily life scenarios (kitchen)</td>
<td align="center">RGB; Speech data</td>
<td align="center">Constructs a Knowledge Graph (KG) that integrates semantic labels, relationships, and properties derived from multimodal data</td>
<td align="center">Using object detection and task goal identification, the robot understands the context and adapts actions to collaborate with human preferences</td>
<td align="center">Knowledge representation and reasoning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B39">Li et al. (2021)</xref>
</td>
<td align="center">A human-robot collaborative assembly (HRCA) robot for aircraft bracket assembly</td>
<td align="center">RGB video; Skeleton data</td>
<td align="center">Intermediate attention captures inter-modality relationships and recalibrates feature maps for better alignment</td>
<td align="center">Recognizing actions and predicting intentions based on visual details and spatiotemporal skeleton data to assist human operators in assembly tasks proactively</td>
<td align="center">Rule-based system</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B10">Belcamino et al. (2024)</xref>
</td>
<td align="center">A multimodal perception framework for human-robot collaboration</td>
<td align="center">RGB-D (Baxter&#x2019;s cameras, Zed2 camera); 4x IMU; Tactaxis Sensor</td>
<td align="center">Outputs from perception modules (vision, touch, IMU) are integrated at the planning stage, with an HTN planner handling task allocation</td>
<td align="center">Assign specific actions to the robot and human based on current states (e.g., object availability, human activity)</td>
<td align="center">Automated planning (NTH)</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B58">Schmidt-Rohr et al. (2008a)</xref>
</td>
<td align="center">A multimodal home service companion robot</td>
<td align="center">RGB-D; Speech (Sphinx4); LiDAR</td>
<td align="center">Raw sensory inputs (speech, human activities, and navigation) are converted into probability distributions</td>
<td align="center">The robot identifies the &#x201c;person of interest&#x201d; based on multimodal cues (speech recognition, human activities) and then autonomously navigates as required</td>
<td align="center">POMDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B9">Baptista et al. (2024)</xref>
</td>
<td align="center">A human-robot collaborative manufacturing unit</td>
<td align="center">RGB-D (2x cam); Force sensor; LiDAR</td>
<td align="center">CNN is used to predict to fuse the operator&#x2019;s next action</td>
<td align="center">Finite State Machine (FSM) transitions between different behaviors (fast object manipulation, safe object handling, object handover) based on multimodal inputs</td>
<td align="center">Rule-based system</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B43">Ly et al. (2024)</xref>
</td>
<td align="center">An interactive home robot</td>
<td align="center">RGB-D; Language</td>
<td align="center">The planner (LLM-based) generates action sequences by integrating recognition results with motion feasibility</td>
<td align="center">The mobile manipulator (Toyota HSR) generates action sequences based on user commands</td>
<td align="center">LLM-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B80">Zhang et al. (2023b)</xref>
</td>
<td align="center">An interactive assistive robot (home use)</td>
<td align="center">RGB; Language</td>
<td align="center">A fine-tuned GPT integrates visual input (object detection results), language input (dialogue), and action decision output</td>
<td align="center">The system generates a robotic operation plan based on detected objects and human instructions</td>
<td align="center">LLM-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B16">Churamani et al. (2020)</xref>
</td>
<td align="center">An affect-driven human-robot interaction robot</td>
<td align="center">RGB (facial); Audio (auditory)</td>
<td align="center">MCCNN network with two independent channels for facial expression recognition and speech emotion recognition, which are then combined into a unified emotional representation</td>
<td align="center">Use reinforcement learning (RL) to train robots for negotiation in the ultimatum game</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B53">Qin et al. (2023)</xref>
</td>
<td align="center">A service robot interaction system for individuals with impaired expressive abilities</td>
<td align="center">Touch; speech; EMG gestures; visual gestures; haptics</td>
<td align="center">Different modalities are combined and integrated at the decision level</td>
<td align="center">Multimodal fusion to recognize user intent and respond accordingly</td>
<td align="center">Rule-based system</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B41">Lu et al. (2023)</xref>
</td>
<td align="center">A vision-language interactive grasping robot.s</td>
<td align="center">RGB-D; Language (RoboRefIt)</td>
<td align="center">A transformer-based cross-modal attention mechanism to fuse extracted features (image, text)</td>
<td align="center">Locate and interactively grasp objects using vision &#x2b; text foundation models and point cloud processing</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B2">Al-Qaderi and Rad (2018b)</xref>
</td>
<td align="center">A multimodal perception system for character recognition in social robots</td>
<td align="center">RGB (FERET); RGB-D (TIDIGITS); RGB-D (3D body and depth)</td>
<td align="center">Spiking neural networks (SNN) process feature vectors from each modality</td>
<td align="center">Social robot character recognition with dynamic selection of the most reliable method</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B60">Scicluna et al. (2024)</xref>
</td>
<td align="center">Robot navigation and hazard avoidance</td>
<td align="center">RGB, LiDAR; IMU</td>
<td align="center">Align 2D bounding boxes extracted from RGB with LiDAR depth data</td>
<td align="center">Multimodal fusion of RGB, event, and LiDAR ensures FP detection does not lead to incorrect decisions</td>
<td align="center">CVFK</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B3">Aly (2014)</xref>
</td>
<td align="center">Robot behavior adapted to human characteristics</td>
<td align="center">RGB; Speech; Text Data</td>
<td align="center">Coupled Hidden Markov Models (CHMMs) are used to model correlations between speech prosody and gesture features</td>
<td align="center">CHMM-driven decisions synthesize synchronized gestures and prosody for naturalistic robot behavior</td>
<td align="center">HMMs</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B86">Zhu et al. (2024)</xref>
</td>
<td align="center">Multimodal emotion recognition simulating the human brain</td>
<td align="center">RGB; Audio; Text Data; (IEMOCAP Dataset)</td>
<td align="center">Pattern fusion with cross-attention mechanism (ACAM)</td>
<td align="center">The cross-attention mechanism for unified multimodal emotion processing effectively utilizes data from different modalities to enhance emotion recognition accuracy</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B84">Zhou and Wachs (2019)</xref>
</td>
<td align="center">Assistant nurse robot (providing medical instruments)</td>
<td align="center">Myo armband; Epoc headset sensor; Kinect sensor</td>
<td align="center">Multimodal signals are synchronized at a 20 Hz frequency and concatenated column-wise</td>
<td align="center">The combination of EEG, EMG, body posture, and acoustic features enables early intent recognition, facilitating predictive decision-making</td>
<td align="center">HMMs</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B48">Nan et al. (2019)</xref>
</td>
<td align="center">Human action recognition on a social assistive robot platform</td>
<td align="center">RGB-D pepper (NTU RGB &#x2b; D Dataset)</td>
<td align="center">Align RGB and depth maps by time frames</td>
<td align="center">Accurate elderly action recognition model based on OpenPose posture estimation module</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B62">Siqueira et al. (2018)</xref>
</td>
<td align="center">Effective emotion recognition under multimodal input conditions</td>
<td align="center">RGB; Speech</td>
<td align="center">Emotion and language models make independent predictions, which are then evaluated using a decision framework</td>
<td align="center">Dynamically resolving emotional mismatches by assessing which modality is more dominant or consistent across multiple interactions</td>
<td align="center">Rule-based system</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B20">Da&#x11f;larl&#x131; (2023)</xref>
</td>
<td align="center">A perception system (game) architecture mimicking human perception</td>
<td align="center">RGB; Audio</td>
<td align="center">Autoencoder used for feature compression</td>
<td align="center">Determining agent actions (escape or fight) through multimodal scene understanding</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B64">Song et al. (2024a)</xref>
</td>
<td align="center">Socially aware robot navigation using vision-language models (VLM)</td>
<td align="center">RGB; LiDAR; Language</td>
<td align="center">GPT-4V</td>
<td align="center">VLM-based approach enables adaptive and socially aware decision-making (e.g., recognizing stop gestures)</td>
<td align="center">VLM-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B63">Siva and Zhang (2022)</xref>
</td>
<td align="center">A method for long-term human following that adapts to ambient lighting variations in real-world environments</td>
<td align="center">RGB-D</td>
<td align="center">Sparse optimization techniques to learn feature weights for each modality</td>
<td align="center">Stable tracking across three different scenarios representing typical lighting variations</td>
<td align="center">MDP</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B26">Ferreira et al. (2012)</xref>
</td>
<td align="center">Hierarchical Bayesian framework for multimodal active humanoid perception</td>
<td align="center">RGB; Audio</td>
<td align="center">Bayesian hierarchical model integrates visual and auditory perception</td>
<td align="center">Bayesian inference mechanism dynamically adjusts perceptual weights of vision and hearing</td>
<td align="center">Bayesian Inference</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B21">Dean-Leon et al. (2017)</xref>
</td>
<td align="center">A bimanual wheeled humanoid robot with multimodal tactile sensing</td>
<td align="center">RGB; Tactile Skin Sensors; LiDAR</td>
<td align="center">Integration at signal, feature, and symbol levels</td>
<td align="center">TOMM integrates tactile skin input with RGB perception to enhance collision avoidance, compliance, and grasping strategies</td>
<td align="center">Symbolic inference</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B52">Peque&#xf1;o-Zurro et al. (2022)</xref>
</td>
<td align="center">A welfare robot guidance system</td>
<td align="center">RGB-D; Lidar; IMU</td>
<td align="center">Human key points and corresponding depth maps are projected into 3D space</td>
<td align="center">Vision &#x2b; depth data estimate human position and velocity, while LiDAR &#x2b; odometry &#x2b; IMU enable safe navigation</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B72">Wu et al. (2019)</xref>
</td>
<td align="center">Robust robotic arm system under external disturbances</td>
<td align="center">Force-torque sensor; Tactile Sensor; RGB-D; Proprioception</td>
<td align="center">Hierarchical Dirichlet Process Hidden Markov Model (HDP-VAR-HMM) captures underlying patterns of different perception modes and probabilistically learns state transitions for fusion</td>
<td align="center">Fault detection, fault diagnosis, and robot task exploration based on multimodal signals</td>
<td align="center">HMMs</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B70">Wang and Feng (2024)</xref>
</td>
<td align="center">Assistive massage robotic arm</td>
<td align="center">RGB-D; Speech</td>
<td align="center">Features (RGB, depth, gestures, speech, and key points) are extracted separately and then integrated into a classifier for decision-making</td>
<td align="center">Predicting the most likely massage points based on multimodal intent recognition</td>
<td align="center">Bayesian method</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B57">Rossi et al. (2024)</xref>
</td>
<td align="center">Bartending robot for long-term interactive operations</td>
<td align="center">RGB; Speech</td>
<td align="center">Merging detailed data from participating modules (emotion from text semantics, speech, and facial expressions)</td>
<td align="center">Real-time analysis of engagement (posture, speech, emotion), personalized and adaptive robot behavior recommendation (dialogue, gestures), and context-aware beverage preparation</td>
<td align="center">Rule-based systems</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B6">Balakuntala et al. (2021)</xref>
</td>
<td align="center">A robot for performing daily contact-rich tasks (cleaning, writing, and peeling)</td>
<td align="center">RGB-D; Tactile; Force-Torque; Joint Position</td>
<td align="center">CNN for RGB-D, MLP for tactile, force, and joint data, followed by a two-layer MLP for feature integration</td>
<td align="center">Perception outputs are used for trajectory execution (writing, cleaning, peeling) via reinforcement learning algorithms</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B75">Yi et al. (2020)</xref>
</td>
<td align="center">Smart home service robot</td>
<td align="center">RGB-D; LiDAR; IMU; Audio; Force-Torque</td>
<td align="center">CNN for RGB-D fusion</td>
<td align="center">Path planning and obstacle avoidance, object grasping and picking, speech command interpretation, and user interaction</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B32">Ivaldi et al. (2013)</xref>
</td>
<td align="center">Humanoid robot active object learning</td>
<td align="center">RGB-D; Sounds; Proprioception; Force/torque</td>
<td align="center">Integrate visual, auditory, and proprioceptive information using incremental learning and sensorimotor loops</td>
<td align="center">Generate an &#x201c;interest map&#x201d; using multimodal sensory data to evaluate different objects and tasks, then determine the most beneficial actions for learning</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B18">Cuay&#xe1;huitl (2020)</xref>
</td>
<td align="center">Interactive multimodal social robot (Tic-Tac-Toe game)</td>
<td align="center">RGB; Speech</td>
<td align="center">Deep reinforcement learning (DQN) model integrates visual perception and speech interaction</td>
<td align="center">The DQN reinforcement learning model integrates multimodal inputs to optimize the robot&#x2019;s action selection strategy</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B45">Mei et al. (2024)</xref>
</td>
<td align="center">Autonomous planning robotic arm with error correction mechanism</td>
<td align="center">RGB-D; Language</td>
<td align="center">GPT-4V to learn a joint representation of vision and language data</td>
<td align="center">Multimodal perception to perform robotic task planning with real-time error correction mechanisms</td>
<td align="center">VLM-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B65">Song et al. (2024b)</xref>
</td>
<td align="center">Indoor navigation and 3D object recognition based on a multimodal knowledge graph</td>
<td align="center">RGB; Language</td>
<td align="center">CLIP-based encoder extracts embeddings from text and images</td>
<td align="center">Link perceptual (visual) data with conceptual (text) knowledge for contextual reasoning</td>
<td align="center">Knowledge representation and reasoning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B29">Granata et al. (2013)</xref>
</td>
<td align="center">Human detection and tracking robot</td>
<td align="center">RGB-D; Laser; Microphones</td>
<td align="center">Information extracted from four detectors is merged using weighted criteria based on the field of view</td>
<td align="center">Extended Kalman Filter (EKF) predicts motion when sensor data is incomplete, enabling robust user action tracking</td>
<td align="center">Probabilistic (EKF)</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B8">Banerjee et al. (2018)</xref>
</td>
<td align="center">The robot can perceive the appropriate timing to interrupt human operations</td>
<td align="center">RGB-D</td>
<td align="center">Euclidean distance heuristics to aggregate the output of the various detectors into a single feature vector</td>
<td align="center">Determine when to interrupt humans based on perceived interruptibility</td>
<td align="center">LDCRF</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B73">Xie and Dames (2023)</xref>
</td>
<td align="center">Autonomous navigation robot for spaces with static obstacles and dense pedestrian traffic</td>
<td align="center">RGB (ZED); LiDAR (Hokuyo); Goal point</td>
<td align="center">Two 80<inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>80 feature maps from images and one 80<inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>80 feature map from LiDAR are combined and fed into a 3<inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>3 Conv2D layer</td>
<td align="center">By fusing LiDAR and pedestrian velocity data, the system predicts pedestrian movement patterns, enabling adaptive speed and direction adjustments</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B17">Cohen-Lhyver et al. (2018)</xref>
</td>
<td align="center">A novel system for regulating robotic exploration behavior</td>
<td align="center">Audio; Visual</td>
<td align="center">M-SOM: where each layer corresponds to audio or video data</td>
<td align="center">The consistency of audiovisual events determines whether to trigger head movement</td>
<td align="center">Knowledge representation and reasoning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B14">Chen et al. (2020)</xref>
</td>
<td align="center">Audiovisual navigation in a 3D environment</td>
<td align="center">Audio; Visual; GPS</td>
<td align="center">Features from vision, audio, and GPS are fed into a GRU.</td>
<td align="center">When the target is occluded, the agent relies more on audio for localization; when avoiding obstacles, the agent relies more on vision for localization</td>
<td align="center">Reinforcement Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B56">Roh (2022)</xref>
</td>
<td align="center">Autonomous driving robot navigation and human-robot interaction</td>
<td align="center">LiDAR; RGB-D</td>
<td align="center">GraphNet-based fusion, encoding states into a graph</td>
<td align="center">Improve the trajectory prediction using interactive compressed topological representations to enable safe and efficient navigation</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B35">Jia et al. (2022)</xref>
</td>
<td align="center">A multimodal emotion recognition model</td>
<td align="center">RGB; Speech; Sequential hand; head action</td>
<td align="center">Three modality features are extracted and fed into an attention layer</td>
<td align="center">Speech (temporal cues), video (spatial facial cues), and motion (body movement cues) to enhance the accuracy and robustness of emotion classification</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B83">Zhao et al. (2024)</xref>
</td>
<td align="center">Audiovisual interaction style recognition for children with autism</td>
<td align="center">Video; Audio</td>
<td align="center">Encoders from video and audio are combined through concatenated encoding</td>
<td align="center">Frame-based video features are combined with synchronized audio and speech signals to enhance the understanding of behavior sequences</td>
<td align="center">Supervised Learning</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B22">Deng et al. (2024)</xref>
</td>
<td align="center">Audiovisual recognition of autism-related behaviors</td>
<td align="center">Video; Audio</td>
<td align="center">Use different feature fusion methods, including weighting, max pooling, attention-based weighting, and concatenation</td>
<td align="center">Recognize and classify autism-related behaviors in video segments</td>
<td align="center">VLM-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B81">Zhang et al. (2024a)</xref>
</td>
<td align="center">Robot planning in open-world environments</td>
<td align="center">RGB; Language</td>
<td align="center">Pretrained VLM (Vision-Language Model)</td>
<td align="center">When an action fails (e.g., grasping a cup fails or an object drops), the system updates the world state and regenerates the plan</td>
<td align="center">VLM &#x2b; PDDL</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B77">Yuan et al. (2024)</xref>
</td>
<td align="center">Assistive social robot for Alzheimer&#x2019;s patients</td>
<td align="center">RGB; Audio; Physiological Signals</td>
<td align="center">Each modality is stored separately, and analysis informs decisions about user performance</td>
<td align="center">Pepper robot provides instructions and evaluates individuals performing ten self-care tasks</td>
<td align="center">Rule-based</td>
</tr>
<tr>
<td align="center">
<xref ref-type="bibr" rid="B38">Lai et al. (2025)</xref>
</td>
<td align="center">Service robot framework in aging societies</td>
<td align="center">Voice commands; Deictic posture; RGB-D</td>
<td align="center">Pretrained LLM (GPT4-Turbo) combines verbal commands (action intent) and deictic postures (object selection) to infer complete human intention</td>
<td align="center">Uses GPT-4 to generate task execution sequences, ensuring collision-free movements</td>
<td align="center">LLM-based</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3-5-1">
<title>3.5.1 Learning-based paradigm</title>
<p>Learning-based decision-making treats the robot (or agent) as a system that acquires policies or value functions from data. Common examples in HRI include supervised learning approaches (e.g., classification, regression), reinforcement learning (RL) for interactive tasks, and imitation learning from human demonstrations. For example, the robot learns to map sensor inputs to discrete actions (e.g., &#x201c;stop,&#x201d; &#x201c;go,&#x201d; &#x201c;turn&#x201d;) based on labeled training sets (<xref ref-type="bibr" rid="B52">Peque&#xf1;o-Zurro et al., 2022</xref>). Similarly, in a collaborative assembly scenario, the robot explores different action strategies, receiving reward signals based on successful or failed assembly interactions (<xref ref-type="bibr" rid="B15">Chen et al., 2025</xref>). Alternatively, the robot can observe a human performing a skill and imitate it, adapting its behavior accordingly. For example (<xref ref-type="bibr" rid="B16">Churamani et al., 2020</xref>), designed an emotion-driven human-robot interaction system using neural network fusion. Their MCCNN (Multi-Channel Convolutional Neural Network) model consists of two independent channels for facial expression recognition and speech emotion recognition, which are then combined into a unified emotional representation. And then, reinforcement learning (RL) is employed to train the robot on negotiation strategies in the ultimatum game. Similarly (<xref ref-type="bibr" rid="B41">Lu et al., 2023</xref>), proposed a vision-language interactive grasping robot, leveraging a transformer-based cross-modal attention mechanism. This system integrates vision, text-based representations, and point cloud processing to enable precise object localization and interactive grasping. Likewise (<xref ref-type="bibr" rid="B2">Al-Qaderi and Rad, 2018b</xref>), utilized network-based fusion via a spiking neural network (SNN) to process feature vectors from multiple modalities. This approach enhances multimodal perception for social robots, enabling dynamic and reliable human recognition by selecting the most robust identification method based on the available sensory data. The core advantages of learning-based decision-making include adaptability to new tasks and improving with more data. However, a key drawback is the potentially large data requirement.</p>
</sec>
<sec id="s3-5-2">
<title>3.5.2 Problem formulation based methods</title>
<p>This approach primarily abstracts decision-making as a mathematical model, such as Markov Decision Processes (MDP), Partially Observable MDPs (POMDP), or game theory. Each formulation represents the agent&#x2019;s state, actions, rewards, and uncertainties. For example (<xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>), employs MOMDPs to integrate multimodal data streams, modeling fully and partially observable state variables with updates from multimodal sensory feedback to optimize future decisions (<xref ref-type="bibr" rid="B5">Amiri et al., 2020</xref>). focuses on learning and reasoning for robot sequential decision-making under uncertainty, using a POMDP planner that leverages sensor data and contextual knowledge as priors to determine the optimal action for proactive HRI. Similarly (<xref ref-type="bibr" rid="B78">Zhang et al., 2021</xref>), applies a dynamically constructed POMDP to fuse information from different sensory modalities and actions to compute the best policy. The decision-making process is driven by the POMDP framework, which refines the robot&#x2019;s belief state using multimodal sensory inputs and determines the optimal course of action. This approach&#x2019;s advantage lies in its clear mathematical framework for problem representation. However, a key drawback is that large state spaces often lead to high computational complexity.</p>
</sec>
<sec id="s3-5-3">
<title>3.5.3 Symbolic/logic-based approaches</title>
<p>This approach relies on symbolic representations (rules, logic programs, knowledge bases) to plan or reason about actions. Examples include rule-based expert systems, automated planning like STRIPS or HTN, and knowledge representation with Answer Set Programming (ASP) or Programming in Logic (Prolog). For example (<xref ref-type="bibr" rid="B23">Diab and Demiris, 2024</xref>), designed an assistive HRI robot for daily life scenarios (kitchen tasks), integrating knowledge-based reasoning to support human-robot collaboration. The system utilizes object detection, spatial awareness, and environmental state assessment (e.g., detecting clutter) through a ROS-integrated version of YOLO trained on the COCO dataset. To enable model-based fusion, the framework constructs a Knowledge Graph (KG) that integrates semantic labels, relationships, and properties derived from multimodal data. By combining object detection and task goal identification, the robot understands the scene context and dynamically adapts its actions to align with human preferences, ensuring more intuitive and effective collaboration. This decision-making approach has the advantages of high-level interpretability, accurate capture of domain knowledge, and logically rigorous reasoning. The drawbacks include poor robustness to noise and potential fragility if the rules are incomplete.</p>
</sec>
<sec id="s3-5-4">
<title>3.5.4 Probabilistic methods</title>
<p>This approach primarily uses Bayesian networks, HMMs, factor graphs, or PGM-based methods to model stochastic processes, particularly for uncertainty in human states or the environment. The system continuously updates probabilities as new observations arrive. For example (<xref ref-type="bibr" rid="B19">Da&#x11f;larl&#x131;, 2020</xref>), employs Bayesian networks for cognitive perception, enabling robots to interact with humans and navigate dynamic environments as personal assistants. Similarly (<xref ref-type="bibr" rid="B84">Zhou and Wachs, 2019</xref>), utilizes HMMs to integrate EEG, EMG, body posture, and acoustic features, allowing early intent recognition for predictive decision-making. Likewise (<xref ref-type="bibr" rid="B3">Aly, 2014</xref>), applies CHMM-driven decision-making to synthesize synchronized gestures and prosody for naturalistic robot behavior. The key advantage of this approach is its principled handling of uncertainty, while the main drawback is the potentially high computational cost for large state spaces.</p>
</sec>
<sec id="s3-5-5">
<title>3.5.5 Search-based planning</title>
<p>Search-Based Planning algorithms (e.g., A&#x2a;, D&#x2a;, MCTS) compute a plan or policy by searching the state or action-space. For example (<xref ref-type="bibr" rid="B36">Khandelwal et al., 2017</xref>), combines probabilistic reasoning and planning (CORPP) to infer missing or ambiguous information, thereby reducing errors in understanding and navigation. The advantage of this approach is efficient exploration and planning in structured environments, while its main drawback is that it does not handle partial observability or complex uncertainty as effectively as POMDPs.</p>
</sec>
<sec id="s3-5-6">
<title>3.5.6 Generative AI decision-making</title>
<p>Generative AI-based decision-making leverages LLMs or VLMs to generate or refine robot actions. The system can dynamically generate responses and action plans by querying a pretrained GPT-like model with a prompt such as &#x201c;Given the environment state, what is the next best action?&#x201c; (<xref ref-type="bibr" rid="B46">Menezes, 2024</xref>).&#x2019;s MuModaR framework integrates multimodal inputs using GIT and GPT-4 for cross-modal alignment of visual, textual, and auditory inputs, enabling real-time feedback-driven decision-making (<xref ref-type="bibr" rid="B43">Ly et al., 2024</xref>). Employs an LLM-based planner to integrate recognition results with motion feasibility, allowing a mobile manipulator (Toyota HSR) to generate action sequences based on user commands (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>). Uses GPT-4V to process visual and context-aware inputs, enabling accurate identification of components and their assembly states (<xref ref-type="bibr" rid="B64">Song D. et al., 2024</xref>). Developed a socially aware robot navigation system leveraging a VLM-based approach (GPT-4V) to allow adaptive and socially aware decision-making (e.g., recognizing a stop gesture). The advantages of this approach include great flexibility and potential zero/few-shot learning capabilities. While large language models are often perceived as less interpretable than classical rule-based systems, recent techniques such as chain-of-thought (<xref ref-type="bibr" rid="B42">Lu et al., 2024</xref>) prompting can improve their transparency and reasoning interpretability. Furthermore, although these models can pose computational challenges in latency-sensitive scenarios, real-time constraints may not be critical in many offline decision-making contexts.</p>
</sec>
<sec id="s3-5-7">
<title>3.5.7 Hybrid approaches</title>
<p>Hybrid Approaches combine two or more methods above&#x2014;for instance, using symbolic rules with a deep RL agent, or using MDP planning plus an LLM to handle high-level language instruction. This method can exploit complementary strengths, e.g., robust uncertainty handling with interpretability. For example (<xref ref-type="bibr" rid="B81">Zhang X. et al., 2024</xref>), developed a robot planning system for open-world environments, leveraging a pre-trained VLM and Planning Domain Definition Language (PDDL) to generate actions. When an action fails (e.g., grasping a cup fails or an object drops), the system updates the world state and replans accordingly. However, integration will be complex.</p>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Building on the previous chapters, we established the necessity of multimodality, highlighted the advantages of multimodal perception in dynamic environments, and summarised how multiple modalities can be fused and integrated into subsequent decision-making processes. We also systematically reviewed the major decision-making methodologies for MPDDM commonly employed in HRI. This section will revisit these findings from a broader perspective, focusing on the current challenges and limitations of existing systems and research efforts. Specifically, we will examine three key areas: technical integration and sensor noise, domain generalization, and safety and robustness. Finally, based on these observations, we propose several future research directions that could guide subsequent investigations and applications in this evolving field.</p>
<sec id="s4-1">
<title>4.1 Current challenges and limitations</title>
<sec id="s4-1-1">
<title>4.1.1 Technical integration and sensor noise</title>
<p>In multimodal perception-driven decision-making (MPDDM) systems, the technical integration of sensors and computational modules remains a significant challenge. First, the need to fuse and align data from multiple modalities (e.g., vision, LiDAR, audio) can introduce high computational complexity&#x2014;the system must handle large-scale data (see (<xref ref-type="bibr" rid="B79">Zhang X. et al., 2023</xref>; <xref ref-type="bibr" rid="B1">Al-Qaderi and Rad, 2018a</xref>; <xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>; <xref ref-type="bibr" rid="B68">Vauf et al., 2016</xref>; <xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>; <xref ref-type="bibr" rid="B67">Tang et al., 2015</xref>; <xref ref-type="bibr" rid="B61">Sha, 2024</xref>; <xref ref-type="bibr" rid="B34">Ji et al., 2020</xref>; <xref ref-type="bibr" rid="B69">Wang, 2023</xref>; <xref ref-type="bibr" rid="B10">Belcamino et al., 2024</xref>)) while ensuring real-time performance. Studies indicate that unimodal perception (using only LiDAR or RGB cameras) is less effective in socially rich or dynamic HRI scenarios (<xref ref-type="bibr" rid="B51">Panigrahi et al., 2023</xref>; <xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>; <xref ref-type="bibr" rid="B5">Amiri et al., 2020</xref>), and highlight the necessity of incorporating multiple sensors (<xref ref-type="bibr" rid="B12">Cai et al., 2024</xref>) for robust situational awareness. However, integrating multiple modalities demands careful calibration, time-stamping, and data synchronization (<xref ref-type="bibr" rid="B61">Sha, 2024</xref>; <xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>).</p>
<p>A second challenge relates to computational complexity and real-time constraints. As multiple modalities scale, so do the demands on both memory and processing power, especially when advanced deep learning is used for sensor fusion (<xref ref-type="bibr" rid="B19">Da&#x11f;larl&#x131;, 2020</xref>; <xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>; <xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>). This inevitably leads to a significant issue&#x2014;the need to sacrifice some degree of precision. This trade-off is one of the key reasons why some studies argue that it is difficult to achieve an accurate representation of the real world. For instance, systems that combine raw images, depth maps, and social signals (e.g., speech or gesture data) can overwhelm onboard hardware if not carefully designed (<xref ref-type="bibr" rid="B61">Sha, 2024</xref>; <xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>; <xref ref-type="bibr" rid="B26">Ferreira et al., 2012</xref>). Consequently, many works struggle to maintain low computational overhead while preserving runtime flexibility and robust decision-making (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>; <xref ref-type="bibr" rid="B36">Khandelwal et al., 2017</xref>; <xref ref-type="bibr" rid="B78">Zhang et al., 2021</xref>; <xref ref-type="bibr" rid="B39">Li et al., 2021</xref>). In addition, some works claimed that handling partial observability (e.g., in a mixed-observability MDP) further intensifies the complexity (<xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>; <xref ref-type="bibr" rid="B34">Ji et al., 2020</xref>).</p>
<p>Finally, sensor noise and environmental uncertainties remain a pervasive obstacle to reliable MPDDM. Vision modules may suffer inaccuracies from changing illumination or strong reflections, and LiDAR scans can be corrupted by cluttered or reflective surfaces (<xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>; <xref ref-type="bibr" rid="B61">Sha, 2024</xref>; <xref ref-type="bibr" rid="B60">Scicluna et al., 2024</xref>). Tactile or audio channels can face similar distortions when interacting closely with humans (e.g., voice overlapping in a crowded environment (<xref ref-type="bibr" rid="B80">Zhang Z. et al., 2023</xref>; <xref ref-type="bibr" rid="B53">Qin et al., 2023</xref>), or haptic signals drowned by mechanical vibration (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>)). While some systems attempt to incorporate uncertainty modeling or real-time sensor re-calibration (<xref ref-type="bibr" rid="B39">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B2">Al-Qaderi and Rad, 2018b</xref>), guaranteeing seamless operation in the presence of sensor noise and incomplete data still remains an open technical challenge.</p>
</sec>
<sec id="s4-1-2">
<title>4.1.2 Domain generalization</title>
<p>Another critical issue for MPDDM in HRI is domain generalization, i.e., whether a trained or engineered system can maintain effectiveness when deployed in new tasks or different application contexts (<xref ref-type="bibr" rid="B79">Zhang X. et al., 2023</xref>; <xref ref-type="bibr" rid="B69">Wang, 2023</xref>; <xref ref-type="bibr" rid="B9">Baptista et al., 2024</xref>; <xref ref-type="bibr" rid="B84">Zhou and Wachs, 2019</xref>; <xref ref-type="bibr" rid="B79">Zhang X. et al., 2023</xref>). For example, in personal-assistive robots, user demographics and cultural factors significantly affect language or gesture recognition modules (<xref ref-type="bibr" rid="B3">Aly, 2014</xref>). Systems that are meticulously tuned to one environment or set of objects often fail to generalize in a new industrial or social setting (<xref ref-type="bibr" rid="B67">Tang et al., 2015</xref>; <xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>), leading to increased development costs each time the context changes.</p>
</sec>
<sec id="s4-1-3">
<title>4.1.3 Adaptation</title>
<p>Adaptation is very important in HRI, Despite promising methods for continual learning, many HRI systems still exhibit limited adaptation to changes in user needs or environmental conditions. Trained models may fail when confronted with new geometries, lighting setups, or people&#x2019;s behaviors (<xref ref-type="bibr" rid="B5">Amiri et al., 2020</xref>; <xref ref-type="bibr" rid="B9">Baptista et al., 2024</xref>; <xref ref-type="bibr" rid="B16">Churamani et al., 2020</xref>; <xref ref-type="bibr" rid="B77">Yuan et al., 2024</xref>). Beyond physical changes, the social nature of HRI demands that systems also account for shifting user preferences, habits, cultural norms, and collaborative task requirements, which can evolve over time (<xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>). Therefore, models optimized for a single/short environment or user profile often become inadequate once real-world conditions diverge from those observed during training. Addressing this challenge calls for robust continual learning strategies that can fuse on-the-fly sensor data with real-time learning/inference while preserving previously acquired knowledge. Developing such flexible adaptation mechanisms remains a key research direction and challenge.</p>
</sec>
<sec id="s4-1-4">
<title>4.1.4 Safety and robustness</title>
<p>Finally, ensuring safety and robustness in real-world HRI scenarios is paramount. Many MPDDM systems must handle close-range human interaction, often in dynamic, unpredictable environments (<xref ref-type="bibr" rid="B46">Menezes, 2024</xref>; <xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>; <xref ref-type="bibr" rid="B74">Yas et al., 2024</xref>). Sensing inaccuracies (e.g., uncertain human motion trajectories or ambiguous gestures) amplify the difficulty of guaranteeing safe robot operation (<xref ref-type="bibr" rid="B67">Tang et al., 2015</xref>), particularly when the robot must execute complex manipulation or navigation tasks (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>). Although advanced approaches leverage multi-layer sensor fusion and failover mechanisms, long-term deployment can still face drift and sensor misalignment (<xref ref-type="bibr" rid="B46">Menezes, 2024</xref>; <xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>), leading to cumulative errors over time.</p>
<p>Moreover, real-time failure recovery is frequently overlooked. Some strategies perform global re-planning upon any anomaly, but this can be computationally expensive or slow (<xref ref-type="bibr" rid="B43">Ly et al., 2024</xref>). Other works rely on users to intervene manually. The challenge is thus twofold: designing motion-level re-planning or fallback strategies without incurring excessive latency, and building a dialogue or feedback loop allowing humans to provide corrective input (<xref ref-type="bibr" rid="B46">Menezes, 2024</xref>; <xref ref-type="bibr" rid="B43">Ly et al., 2024</xref>). All these works aim to create a system that not only meets the real-time demands of dynamic HRI but also maintains robust performance and safe collaborative interactions.</p>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 Future research directions</title>
<sec id="s4-2-1">
<title>4.2.1 Advancing learning-based approaches</title>
<p>Future work would underscore the need for more efficient learning paradigms&#x2014;ranging from generative AI to reinforcement learning (RL)&#x2014;to cope with multimodal perception in dynamic HRI. Several works point to semi-supervised or weakly supervised techniques that can reduce reliance on extensive labeled data, thus lowering the costs of large-scale multimodal curation (<xref ref-type="bibr" rid="B12">Cai et al., 2024</xref>; <xref ref-type="bibr" rid="B51">Panigrahi et al., 2023</xref>; <xref ref-type="bibr" rid="B79">Zhang X. et al., 2023</xref>; <xref ref-type="bibr" rid="B61">Sha, 2024</xref>). Furthermore, robust generative models could help unify multiple input streams (e.g., vision, audio, haptics) while automatically aligning them with latent representations (<xref ref-type="bibr" rid="B19">Da&#x11f;larl&#x131;, 2020</xref>; <xref ref-type="bibr" rid="B46">Menezes, 2024</xref>; <xref ref-type="bibr" rid="B68">Vauf et al., 2016</xref>). Equally important is the push to incorporate advanced attention mechanisms and semantic reasoning for better capturing cross-modal signals (<xref ref-type="bibr" rid="B19">Da&#x11f;larl&#x131;, 2020</xref>; <xref ref-type="bibr" rid="B34">Ji et al., 2020</xref>; <xref ref-type="bibr" rid="B74">Yas et al., 2024</xref>; <xref ref-type="bibr" rid="B3">Aly, 2014</xref>), leading to more context-aware and &#x201c;cognitive&#x201d; HRI systems. In parallel, scaling up sensor coverage while keeping memory overhead tractable remains a challenge that future work must address by optimizing sensor fusion and feature extraction (<xref ref-type="bibr" rid="B68">Vauf et al., 2016</xref>; <xref ref-type="bibr" rid="B10">Belcamino et al., 2024</xref>; <xref ref-type="bibr" rid="B60">Scicluna et al., 2024</xref>).</p>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Improving explainability and human trust</title>
<p>Ensuring that AI-driven decisions are interpretable is vital for fostering user acceptance and trust in HRI (<xref ref-type="bibr" rid="B44">Mathur et al., 2025</xref>). Although many multimodal models yield high accuracy, they often behave as &#x201c;black boxes,&#x201d; making it unclear why a system chooses a particular action or how it handles ambiguous inputs (<xref ref-type="bibr" rid="B67">Tang et al., 2015</xref>; <xref ref-type="bibr" rid="B53">Qin et al., 2023</xref>). Work in (<xref ref-type="bibr" rid="B13">Ch et al., 2022</xref>; <xref ref-type="bibr" rid="B39">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B80">Zhang Z. et al., 2023</xref>) highlights the importance of building affective and dialogue-based interactions with the user, so the robot can clarify uncertainties or explain the reasoning behind certain decisions. Additionally, to promote safer collaboration (<xref ref-type="bibr" rid="B43">Ly et al., 2024</xref>; <xref ref-type="bibr" rid="B41">Lu et al., 2023</xref>), propose integrated motion-level re-planning frameworks that combine transparency about failure causes (e.g., &#x201c;object not in view&#x201d;) with real-time user feedback. Future research might incorporate social cues such as facial expressions, body posture, or personal preferences (<xref ref-type="bibr" rid="B68">Vauf et al., 2016</xref>; <xref ref-type="bibr" rid="B16">Churamani et al., 2020</xref>; <xref ref-type="bibr" rid="B2">Al-Qaderi and Rad, 2018b</xref>; <xref ref-type="bibr" rid="B26">Ferreira et al., 2012</xref>; <xref ref-type="bibr" rid="B77">Yuan et al., 2024</xref>), improving the system&#x2019;s ability to provide human-readable justifications and adapt its behavior accordingly. Ultimately, bridging model decisions and intuitive explanations can drive deeper user trust in situations that demand joint decision-making.</p>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Scalable multi-robot collaboration</title>
<p>Another promising direction concerns scalable multi-robot systems, where tasks span collaborative assembly, multi-robot coordination, or large-scale monitoring (<xref ref-type="bibr" rid="B36">Khandelwal et al., 2017</xref>; <xref ref-type="bibr" rid="B78">Zhang et al., 2021</xref>). While single-robot multimodal perception has progressed substantially, simultaneously coordinating multiple robots under uncertain or partially observable conditions remains underexplored. Key open questions center on robust joint perception&#x2014;sharing or transferring learned policies, sensorimotor features, and knowledge across heterogeneous platforms (<xref ref-type="bibr" rid="B78">Zhang et al., 2021</xref>). In parallel, the complexities of real-world scheduling, path planning, and dynamic role assignment amplify in multi-robot teams, as partial failures in one platform can cascade. Interweaving user interactions&#x2014;e.g., a human operator or supervisor who provides on-demand clarifications&#x2014;poses further integration challenges (<xref ref-type="bibr" rid="B36">Khandelwal et al., 2017</xref>). Addressing these issues could enable more flexible, self-organized teams of robots that better adapt to large-scale tasks and diverse users.</p>
</sec>
<sec id="s4-2-4">
<title>4.2.4 Long-term autonomy and continual learning</title>
<p>Finally, long-term autonomy in dynamic human environments demands that a robot continuously refine its models and maintain stable performance over lengthy deployments (<xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>; <xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>; <xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>). Systems must confront persistent changes in environment geometry, lighting conditions, or occupant behavior, which can degrade originally trained models (<xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>). Continual learning approaches that leverage streaming sensor data could keep the robot&#x2019;s perception and action policies up-to-date&#x2014;though care must be taken to avoid catastrophic forgetting (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>). Equally important is capturing evolving user preferences, social context, and task requirements (<xref ref-type="bibr" rid="B82">Zhang Y. et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Granata et al., 2012</xref>). Achieving robust online updates for these modules will demand balancing data quality (potentially from incomplete or noisy real-world streams) with computational efficiency, as pointed out by (<xref ref-type="bibr" rid="B4">Amiri et al., 2018</xref>) and (<xref ref-type="bibr" rid="B27">Forlini et al., 2024</xref>). Future work may combine online transfer learning, policy gradient RL, and environment mapping methods to sustain consistent performance in long-duration, continuously changing settings.</p>
<p>Overall, addressing these four broad directions&#x2014;advanced learning paradigms, intuitive human trust, multi-robot scaling collaboration, and long-term autonomy&#x2014;holds the potential to push MPDDM-based HRI toward more natural, capable, adaptable, and user-aligned HRI systems in real-world practice.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this survey, we have examined how multimodal perception can enrich decision-making in human-robot interaction from several application perspectives, thereby demonstrating the importance of multimodality in improving decision-making. By synthesizing insights from existing literature, we showed that leveraging multiple sensory modalities not only increases robustness against sensor failures and environmental uncertainties but also provides a richer context for understanding human states and intentions. Consequently, effectively fusing different data modalities into decision models that handle partial observability, real-time constraints, and evolving user behavior has emerged as a critical direction in achieving natural, robust, and safe human-robot interaction in the future.</p>
<p>Despite these promising developments, several challenges remain. Real-world deployments still grapple with sensor noise, synchronization overhead, and the substantial computational burden of processing large-scale multimodal data in real time. Moreover, generalizing systems beyond controlled laboratory conditions poses considerable difficulties&#x2014;especially when robots operate in diverse settings with varied user profiles, tasks, and cultural norms. Safety and trustworthiness also demand deeper investigation; while fusion-based models achieve higher accuracy, they can be opaque, making it difficult for end users/researchers to understand how a robot arrives at particular choices.</p>
<p>Looking ahead, the ongoing progress of learning-based methods and large-scale foundational models is poised to broaden the horizons of what multimodal perception and decision-making can accomplish. By striking a careful balance among computational efficiency, explainability, and responsiveness, future research can produce truly adaptive, socially aware robots that seamlessly integrate into daily life. Ultimately, overcoming these human-centered challenges will bring us closer to robots capable of robustly perceiving complex scenarios, inferring user intentions and needs, and collaborating safely and intelligently across a wide range of domains.</p>
</sec>
</body>
<back>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>WZ: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Project administration, Resources, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. KG: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Writing &#x2013; review and editing. FY: Conceptualization, Data curation, Investigation, Methodology, Project administration, Supervision, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<ack>
<p>The authors acknowledge the use of ChatGPT for editing and polishing the manuscript.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. ChatGPT was used for editing the authors, own text to polish the manuscript (improve the readability).</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Qaderi</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Rad</surname>
<given-names>A. B.</given-names>
</name>
</person-group> (<year>2018a</year>). <article-title>A brain-inspired multi-modal perceptual system for social robots: an experimental realization</article-title>. <source>IEEE Access</source> <volume>6</volume>, <fpage>35402</fpage>&#x2013;<lpage>35424</lpage>. <pub-id pub-id-type="doi">10.1109/access.2018.2851841</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Qaderi</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Rad</surname>
<given-names>A. B.</given-names>
</name>
</person-group> (<year>2018b</year>). <article-title>A multi-modal person recognition system for social robots</article-title>. <source>Appl. Sci.</source> <volume>8</volume>, <fpage>387</fpage>. <pub-id pub-id-type="doi">10.3390/app8030387</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Aly</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <source>
<italic>Towards an interactive human-robot relationship: Developing a customized robot Behavior to human profile.</italic> Ph.D. Thesis</source>. <publisher-loc>Palaiseau, France</publisher-loc>: <publisher-name>ENSTA ParisTech</publisher-name>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Amiri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sinapov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Thomason</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Robot behavioral exploration and multi-modal perception using dynamically constructed controllers</article-title>,&#x201d; in <source>2018 AAAI spring symposium Series</source>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amiri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shirazi</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Learning and reasoning for robot sequential decision making under uncertainty</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>34</volume>, <fpage>2726</fpage>&#x2013;<lpage>2733</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v34i03.5659</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Balakuntala</surname>
<given-names>M. V.</given-names>
</name>
<name>
<surname>Kaur</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wachs</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Voyles</surname>
<given-names>R. M.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Learning multimodal contact-rich skills from demonstrations without reward engineering</article-title>,&#x201d; in <source>2021 IEEE international conference on robotics and automation (ICRA)</source> (<publisher-name>IEEE</publisher-name>), <fpage>4679</fpage>&#x2013;<lpage>4685</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baltru&#x161;aitis</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ahuja</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Morency</surname>
<given-names>L.-P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Multimodal machine learning: a survey and taxonomy</article-title>. <source>IEEE Trans. pattern analysis Mach. Intell.</source> <volume>41</volume>, <fpage>423</fpage>&#x2013;<lpage>443</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2018.2798607</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banerjee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Feigh</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chernova</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Effects of interruptibility-aware robot behavior</article-title>. <source>arXiv Prepr. arXiv:1804.06383</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1804.06383</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baptista</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Castro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gomes</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Amaral</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Human&#x2013;robot collaborative manufacturing cell with learning-based interaction abilities</article-title>. <source>Robotics</source> <volume>13</volume>, <fpage>107</fpage>. <pub-id pub-id-type="doi">10.3390/robotics13070107</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Belcamino</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kilina</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lastrico</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Carf&#xec;</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mastrogiovanni</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A modular framework for flexible planning in human-robot collaboration</article-title>. <source>arXiv Prepr. arXiv:2406.04907</source>, <fpage>2303</fpage>&#x2013;<lpage>2310</lpage>. <pub-id pub-id-type="doi">10.1109/ro-man60168.2024.10731451</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bonci</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cen Cheng</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Indri</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nabissi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sibona</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Human-robot perception in industrial environments: a survey</article-title>. <source>Sensors</source> <volume>21</volume>, <fpage>1571</fpage>. <pub-id pub-id-type="doi">10.3390/s21051571</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>The multi-modal robot perception, language information, and environment prediction model based on deep learning</article-title>. <source>J. Organ. End User Comput. (JOEUC)</source> <volume>36</volume>, <fpage>1</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.4018/joeuc.349987</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Churamani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Barros</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gunes</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wermter</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Affect-driven learning of robot behaviour for collaborative human-robot interactions</article-title>. <source>Front. Robotics AI</source> <volume>9</volume>, <fpage>717193</fpage>. <pub-id pub-id-type="doi">10.3389/frobt.2022.717193</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Schissler</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gari</surname>
<given-names>S. V. A.</given-names>
</name>
<name>
<surname>Al-Halah</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ithapu</surname>
<given-names>V. K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Soundspaces: audio-visual navigation in 3d environments</article-title>,&#x201d; in <source>Computer vision&#x2013;ECCV 2020: 16th European conference, Glasgow, UK, August 23&#x2013;28, 2020, Proceedings, Part VI 16</source> (<publisher-name>Springer</publisher-name>), <fpage>17</fpage>&#x2013;<lpage>36</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Knowledge sharing-enabled low-code program for collaborative robots in mix-model assembly</article-title>. <source>J. Industrial Inf. Integration</source> <volume>45</volume>, <fpage>100824</fpage>. <pub-id pub-id-type="doi">10.1016/j.jii.2025.100824</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Churamani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Barros</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gunes</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wermter</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Affect-driven modelling of robot personality for collaborative human-robot interactions</article-title>. <source>arXiv Prepr. arXiv:2010.07221</source>. . <pub-id pub-id-type="doi">10.48550/arXiv.2010.07221</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cohen-Lhyver</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Argentieri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gas</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The head turning modulation system: an active multimodal paradigm for intrinsically motivated exploration of unknown environments</article-title>. <source>Front. Neurorobotics</source> <volume>12</volume>, <fpage>60</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2018.00060</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cuay&#xe1;huitl</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A data-efficient deep learning approach for deployable multimodal social robots</article-title>. <source>Neurocomputing</source> <volume>396</volume>, <fpage>587</fpage>&#x2013;<lpage>598</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2018.09.104</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Da&#x11f;larl&#x131;</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A cognitive integrated multi-modal perception mechanism and dynamic world modeling for social robot assistants</article-title>. <source>J. Cognitive Syst.</source> <volume>5</volume>, <fpage>46</fpage>&#x2013;<lpage>50</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Da&#x11f;larl&#x131;</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Design of the integrated cognitive perception model for developing situation-awareness of an autonomous smart agent</article-title>. <source>Balkan J. Electr. Comput. Eng.</source> <volume>11</volume>, <fpage>283</fpage>&#x2013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.17694/bajece.1310607</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dean-Leon</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pierce</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bergner</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mittendorfer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ramirez-Amaro</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Burger</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>Tomm: tactile omnidirectional mobile manipulator</article-title>,&#x201d; in <source>2017 IEEE international conference on robotics and automation (ICRA)</source> (<publisher-name>IEEE</publisher-name>), <fpage>2441</fpage>&#x2013;<lpage>2447</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kosloski</surname>
<given-names>E. E.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Barnett</surname>
<given-names>Z. A.</given-names>
</name>
<name>
<surname>Nan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kaplan</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Hear me, see me, understand me: audio-visual autism behavior recognition</article-title>. <source>arXiv Prepr. arXiv:2406.02554</source>. <pub-id pub-id-type="doi">10.1109/TMM.2024.3521838</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diab</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Demiris</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A framework for trust-related knowledge transfer in human&#x2013;robot interaction</article-title>. <source>Aut. Agents Multi-Agent Syst.</source> <volume>38</volume>, <fpage>24</fpage>. <pub-id pub-id-type="doi">10.1007/s10458-024-09653-w</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhuang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Multimodal perception-fusion-control and human&#x2013;robot collaboration in manufacturing: a review</article-title>. <source>Int. J. Adv. Manuf. Technol.</source> <volume>132</volume>, <fpage>1071</fpage>&#x2013;<lpage>1093</lpage>. <pub-id pub-id-type="doi">10.1007/s00170-024-13385-2</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duncan</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Alambeigi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pryor</surname>
<given-names>M. W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A survey of multimodal perception methods for human-robot interaction in social environments</article-title>. <source>ACM Trans. Human-Robot Interact.</source> <volume>13</volume>, <fpage>1</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1145/3657030</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferreira</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Tsiourti</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dias</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Learning emergent behaviours for a hierarchical bayesian framework for active robotic perception</article-title>. <source>Cogn. Process.</source> <volume>13</volume>, <fpage>155</fpage>&#x2013;<lpage>159</lpage>. <pub-id pub-id-type="doi">10.1007/s10339-012-0481-9</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Forlini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Babcinschi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Palmieri</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Neto</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>D-rmgpt: robot-assisted collaborative tasks driven by large multimodal models</article-title>. <source>arXiv Prepr. arXiv:2408.11761</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2408.11761</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Granata</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bidaud</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chetouani</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Melchior</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Multimodal human detection and fuzzy decisional engine for interactive behaviors of a mobile robot</article-title>,&#x201d; in <source>2012 IEEE 3rd international conference on cognitive Infocommunications (CogInfoCom)</source> (<publisher-name>IEEE</publisher-name>), <fpage>395</fpage>&#x2013;<lpage>400</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Granata</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bidaud</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Salini</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ady</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Human activity analysis: a personal robot integrating a framework for robust person detection and tracking and physical based motion analysis</article-title>. <source>Paladyn, J. Behav. Robotics</source> <volume>4</volume>, <fpage>131</fpage>&#x2013;<lpage>146</lpage>. <pub-id pub-id-type="doi">10.2478/pjbr-2013-0011</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grigorescu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Trasnea</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cocias</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Macesanu</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A survey of deep learning techniques for autonomous driving</article-title>. <source>J. field robotics</source> <volume>37</volume>, <fpage>362</fpage>&#x2013;<lpage>386</lpage>. <pub-id pub-id-type="doi">10.1002/rob.21918</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>What makes multi-modal learning better than single (provably)</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>34</volume>, <fpage>10944</fpage>&#x2013;<lpage>10956</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2106.04538</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ivaldi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lyubova</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Droniou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Padois</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Filliat</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Oudeyer</surname>
<given-names>P.-Y.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Object learning through active exploration</article-title>. <source>IEEE Trans. Aut. Ment. Dev.</source> <volume>6</volume>, <fpage>56</fpage>&#x2013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1109/TAMD.2013.2280614</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jahanmahin</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Masoud</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rickli</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Djuric</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Human-robot interactions in manufacturing: a survey of human behavior modeling</article-title>. <source>Robotics Computer-Integrated Manuf.</source> <volume>78</volume>, <fpage>102404</fpage>. <pub-id pub-id-type="doi">10.1016/j.rcim.2022.102404</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Towards shared autonomy framework for human-aware motion planning in industrial human-robot collaboration</article-title>,&#x201d; in <source>2020 IEEE 16th international conference on automation Science and engineering (CASE)</source> (<publisher-name>IEEE</publisher-name>), <fpage>411</fpage>&#x2013;<lpage>417</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A multimodal emotion recognition model integrating speech, video and mocap</article-title>. <source>Multimedia Tools Appl.</source> <volume>81</volume>, <fpage>32265</fpage>&#x2013;<lpage>32286</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-022-13091-9</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khandelwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sinapov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Leonetti</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thomason</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Bwibots: a platform for bridging the gap between ai and human&#x2013;robot interaction research</article-title>. <source>Int. J. Robotics Res.</source> <volume>36</volume>, <fpage>635</fpage>&#x2013;<lpage>659</lpage>. <pub-id pub-id-type="doi">10.1177/0278364916688949</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Pertsch</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Karamcheti</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Balakrishna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Openvla: an open-source vision-language-action model</article-title>. <source>arXiv Prepr. arXiv:2406.09246</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2406.09246</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nassar</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Weber</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>R&#xe4;tsch</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Nvp-hri: zero shot natural voice and posture-based human&#x2013;robot interaction via large language model</article-title>. <source>Expert Syst. Appl.</source> <volume>268</volume>, <fpage>126360</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2024.126360</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Toward proactive human&#x2013;robot collaborative assembly: a multimodal transfer-learning-enabled action prediction approach</article-title>. <source>IEEE Trans. Industrial Electron.</source> <volume>69</volume>, <fpage>8579</fpage>&#x2013;<lpage>8588</lpage>. <pub-id pub-id-type="doi">10.1109/tie.2021.3105977</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Zadeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Morency</surname>
<given-names>L.-P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Foundations and trends in multimodal machine learning: Principles, challenges, and open questions</article-title>. <source>arXiv Prepr. arXiv:2209.03430</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2209.03430</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Vl-grasp: a 6-dof interactive grasp policy for language-oriented objects in cluttered indoor scenes</source>. <publisher-name>IEEE/RSJ International Conference on Intelligent Robots and Systems IROS</publisher-name>, <fpage>976</fpage>&#x2013;<lpage>983</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Deepseek-vl: towards real-world vision-language understanding</article-title>. <source>arXiv Prepr. arXiv:2403.05525</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2403.05525</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ly</surname>
<given-names>K. T.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Havoutis</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Inteliplan: interactive lightweight llm-based planner for domestic robot autonomy</article-title>. <source>arXiv Prepr. arXiv:2409.14506</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2409.14506</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mathur</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Morency</surname>
<given-names>L.-P.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Social genome</article-title>. <source>Grounded Soc. Reason. Abil. multimodal models</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2502.15109</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mei</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>G.-N.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Replanvlm: Replanning robotic tasks with visual language models</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume>, <fpage>10201</fpage>&#x2013;<lpage>10208</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2024.3471457</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Menezes</surname>
<given-names>J. C.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Mumodar: multi-modal framework for human-robot collaboration in cyber-physical systems</article-title>,&#x201d; in <source>Companion of the 2024 ACM/IEEE international conference on human-robot interaction</source>, <fpage>755</fpage>&#x2013;<lpage>759</lpage>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moher</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liberati</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tetzlaff</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Preferred reporting items for systematic reviews and meta-analyses: the prisma statement</article-title>. <source>Ann. Intern. Med.</source> <volume>151</volume>, <fpage>264</fpage>&#x2013;<lpage>269</lpage>. <pub-id pub-id-type="doi">10.7326/0003-4819-151-4-200908180-00135</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ghi&#x21b;&#x103;</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Gavril</surname>
<given-names>A.-F.</given-names>
</name>
<name>
<surname>Trascau</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sorici</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cramariuc</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>Human action recognition for social robots</article-title>,&#x201d; in <source>2019 22nd international conference on Control systems and computer Science (CSCS)</source> (<publisher-name>IEEE</publisher-name>), <fpage>675</fpage>&#x2013;<lpage>681</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<collab>OpenAI</collab> (<year>2023</year>). <article-title>Gpt-4v(ision) system card</article-title>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pandey</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Gelin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A Mass-Produced sociable humanoid robot: Pepper: the first machine of its Kind</article-title>. <source>IEEE Robotics &#x26; Automation Mag.</source> <volume>25</volume>, <fpage>40</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1109/mra.2018.2833157</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Panigrahi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Raj</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Nazeri</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A study on learning social robot navigation with multimodal perception</article-title>. <source>arXiv Prepr. arXiv:2309.12568</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2309.12568</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peque&#xf1;o-Zurro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ignasov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ram&#xed;rez</surname>
<given-names>E. R.</given-names>
</name>
<name>
<surname>Haarslev</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Juel</surname>
<given-names>W. K.</given-names>
</name>
<name>
<surname>Bodenhagen</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Proactive control for online individual user adaptation in a welfare robot guidance scenario: toward supporting elderly people</article-title>. <source>IEEE Trans. Syst. man, Cybern. Syst.</source> <volume>53</volume>, <fpage>3364</fpage>&#x2013;<lpage>3376</lpage>. <pub-id pub-id-type="doi">10.1109/TSMC.2022.3224366</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qin</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A multimodal domestic service robot interaction system for people with declined abilities to express themselves</article-title>. <source>Intell. Serv. Robot.</source> <volume>16</volume>, <fpage>373</fpage>&#x2013;<lpage>392</lpage>. <pub-id pub-id-type="doi">10.1007/s11370-023-00466-6</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reimann</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Kunneman</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Oertel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hindriks</surname>
<given-names>K. V.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A survey on dialogue management in human-robot interaction</article-title>. <source>ACM Trans. Human-Robot Interact.</source> <volume>13</volume>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1145/3648605</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robinson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tidd</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Campbell</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kuli&#x107;</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Corke</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Robotic vision for human-robot interaction and collaboration: a survey and systematic review</article-title>. <source>ACM Trans. Human-Robot Interact.</source> <volume>12</volume>, <fpage>1</fpage>&#x2013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1145/3570731</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Roh</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Approaches for interactions in robotics applications</source>. <publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>University of Washington</publisher-name>.</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rossi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rossi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Maro</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Origlia</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Brillo: Personalised hri with a bartender robot</article-title>
</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schmidt-Rohr</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Knoop</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>L&#xf6;sch</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dillmann</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2008a</year>). &#x201c;<article-title>A probabilistic control architecture for robust autonomy of an anthropomorphic service robot</article-title>,&#x201d; in <source>International conference on cognitive systems</source>. <comment>
<italic>Karlsruhe, Germany</italic>
</comment>.</citation>
</ref>
<ref id="B59">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schmidt-Rohr</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Knoop</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>L&#xf6;sch</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dillmann</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2008b</year>). &#x201c;<article-title>Reasoning for a multi-modal service robot considering uncertainty in human-robot interaction</article-title>,&#x201d; in <source>Proceedings of the 3rd ACM/IEEE international conference on Human robot interaction</source>, <fpage>249</fpage>&#x2013;<lpage>254</lpage>.</citation>
</ref>
<ref id="B60">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Scicluna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Le Gentil</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sutjipto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Paul</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Towards robust perception for assistive robotics: an rgb-event-lidar dataset and multi-modal detection pipeline</article-title>,&#x201d; in <source>2024 IEEE 20th international conference on automation Science and engineering (CASE)</source> (<publisher-name>IEEE</publisher-name>), <fpage>920</fpage>&#x2013;<lpage>925</lpage>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sha</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Multimodal perception system for real open environment</article-title>. <source>arXiv Prepr. arXiv:2410.07926</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2410.07926</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Siqueira</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sutherland</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Barros</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kerzel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Magg</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wermter</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Disambiguating affective stimulus associations for robot perception and dialogue</article-title>,&#x201d; in <source>2018 IEEE-RAS 18th international conference on humanoid robots (Humanoids)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siva</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Robot perceptual adaptation to environment changes for long-term human teammate following</article-title>. <source>Int. J. Robotics Res.</source> <volume>41</volume>, <fpage>706</fpage>&#x2013;<lpage>720</lpage>. <pub-id pub-id-type="doi">10.1177/0278364919896625</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Payandeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Manocha</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024a</year>). <article-title>Socially aware robot navigation through scoring using vision-language models</article-title>. <source>arXiv Prepr. arXiv:2404.00210</source>. <pub-id pub-id-type="doi">10.1109/LRA.2024.3511409</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024b</year>). <article-title>Scene-driven multimodal knowledge graph construction for embodied ai</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>36</volume>, <fpage>6962</fpage>&#x2013;<lpage>6976</lpage>. <pub-id pub-id-type="doi">10.1109/tkde.2024.3399746</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sandoval</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Laribi</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Recent advancements in multimodal human&#x2013;robot interaction</article-title>. <source>Front. Neurorobotics</source> <volume>17</volume>, <fpage>1084000</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2023.1084000</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yusuf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Botzheim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kubota</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>C. S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A novel multimodal communication framework using robot partner for aging population</article-title>. <source>Expert Syst. Appl.</source> <volume>42</volume>, <fpage>4540</fpage>&#x2013;<lpage>4555</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2015.01.016</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaufreydaz</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Johal</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Combe</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Starting engagement detection towards a companion robot using multimodal features</article-title>. <source>Robotics Aut. Syst.</source> <volume>75</volume>, <fpage>4</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1016/j.robot.2015.01.004</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Res-flnet: human-robot interaction and collaboration for multi-modal sensing robot autonomous driving tasks based on learning control algorithm</article-title>,&#x201d;. <source>Front. Neurorobot</source> <volume>17</volume> <fpage>1269105</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2023.1269105</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Multimodal massage localization algorithm for human acupoints</article-title>. <source>Int. J. Human&#x2013;Computer Interact.</source> <volume>41</volume>, <fpage>6011</fpage>&#x2013;<lpage>6028</lpage>. <pub-id pub-id-type="doi">10.1080/10447318.2024.2372893</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Multimodal human&#x2013;robot interaction for human-centric smart manufacturing: a survey</article-title>. <source>Adv. Intell. Syst.</source> <volume>6</volume>, <fpage>2300359</fpage>. <pub-id pub-id-type="doi">10.1002/aisy.202300359</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Incremental learning introspective movement primitives from multimodal unstructured demonstrations</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>159022</fpage>&#x2013;<lpage>159036</lpage>. <pub-id pub-id-type="doi">10.1109/access.2019.2947529</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dames</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Drl-vo: learning to navigate through crowded dynamic scenes using velocity obstacles</article-title>. <source>IEEE Trans. Robotics</source> <volume>39</volume>, <fpage>2700</fpage>&#x2013;<lpage>2719</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2023.3257549</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yasar</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Islam</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Iqbal</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Imprint: Interactional dynamics-aware motion prediction in teams using multimodal context</article-title>. <source>ACM Trans. Human-Robot Interact.</source> <volume>13</volume>, <fpage>1</fpage>&#x2013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1145/3626954</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yi</surname>
<given-names>J.-B.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>S.-J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Unified software platform for intelligent home service robots</article-title>. <source>Appl. Sci.</source> <volume>10</volume>, <fpage>5874</fpage>. <pub-id pub-id-type="doi">10.3390/app10175874</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Robot behavior generation and human behavior understanding in natural human-robot interaction</source>. <publisher-loc>Paris</publisher-loc>: <publisher-name>Institut Polytechnique de</publisher-name>. <comment>Ph.D. thesis</comment>.</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bray</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Oliver</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Duzan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Crane</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A social robot-facilitated performance assessment of self-care skills for people with alzheimer&#x2019;s: a preliminary study</article-title>. <source>Int. J. Soc. Robotics</source> <volume>16</volume>, <fpage>2065</fpage>&#x2013;<lpage>2078</lpage>. <pub-id pub-id-type="doi">10.1007/s12369-024-01174-6</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sinapov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Planning multimodal exploratory actions for online robot attribute learning</article-title>. <source>arXiv Prepr. arXiv:2106.03029</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2106.03029</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Amiri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sinapov</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Thomason</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Multimodal embodied attribute learning by robots for object-centric action policies</article-title>. <source>Aut. Robots</source> <volume>47</volume>, <fpage>505</fpage>&#x2013;<lpage>528</lpage>. <pub-id pub-id-type="doi">10.1007/s10514-023-10098-5</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Mani-gpt: a generative model for interactive robotic manipulation</article-title>. <source>Procedia Comput. Sci.</source> <volume>226</volume>, <fpage>149</fpage>&#x2013;<lpage>156</lpage>. <pub-id pub-id-type="doi">10.1016/j.procs.2023.10.649</pub-id>
</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Altaweel</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hayamizu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Amiri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). <article-title>Dkprompt: domain knowledge prompting vision-language models for open-world planning</article-title>. <source>arXiv Prepr. arXiv:2406.17659</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2406.17659</pub-id>
</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>Multimodal perception for indoor mobile robotics navigation and safe manipulation</article-title>. <source>IEEE Trans. Cognitive Dev. Syst.</source>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1109/tcds.2024.3481457</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>K.-M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>C. H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Av-fos: a transformer-based audio-visual interaction style recognition for children with autism based on the family observation schedule (fos-ii)</article-title>. <source>Authorea Prepr</source>. <pub-id pub-id-type="doi">10.1109/JBHI.2025.3542066</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wachs</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Spiking neural networks for early prediction in human&#x2013;robot collaboration</article-title>. <source>Int. J. Robotics Res.</source> <volume>38</volume>, <fpage>1619</fpage>&#x2013;<lpage>1643</lpage>. <pub-id pub-id-type="doi">10.1177/0278364919872252</pub-id>
</citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Chatvla: unified multimodal understanding and robot control with vision-language-action model</article-title>. <source>arXiv Prepr. arXiv:2502</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2502.14420</pub-id>
</citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Emotion recognition based on brain-like multimodal hierarchical perception</article-title>. <source>Multimedia Tools Appl.</source> <volume>83</volume>, <fpage>56039</fpage>&#x2013;<lpage>56057</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-023-17347-w</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>