<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title-group>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1528754</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1528754</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>A multi-user multi-robot multi-goal multi-device human-robot interaction manipulation benchmark</article-title>
<alt-title alt-title-type="left-running-head">Yoshida et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1528754">10.3389/frobt.2025.1528754</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Yoshida</surname>
<given-names>Akito</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2905849"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Dossa</surname>
<given-names>Rousslan Fernand Julien</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2069821"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Di Vincenzo</surname>
<given-names>Marina</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<uri xlink:href="https://loop.frontiersin.org/people/2723265"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sujit</surname>
<given-names>Shivakanth</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<uri xlink:href="https://loop.frontiersin.org/people/3109518"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Douglas</surname>
<given-names>Hannah</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Arulkumaran</surname>
<given-names>Kai</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2295365"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>Araya Inc.</institution>, <city>Tokyo</city>, <country country="JP">Japan</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Kai Arulkumaran, <email xlink:href="kai_arulkumaran@araya.org">kai_arulkumaran@araya.org</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>&#x2020;</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-09-25">
<day>25</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1528754</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Yoshida, Dossa, Di Vincenzo, Sujit, Douglas and Arulkumaran.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Yoshida, Dossa, Di Vincenzo, Sujit, Douglas and Arulkumaran</copyright-holder>
<license>
<ali:license_ref start_date="2025-09-25">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>One weakness of human-robot interaction (HRI) research is the lack of reproducible results, due to the lack of standardised benchmarks. In this work we introduce a multi-user multi-robot multi-goal multi-device manipulation benchmark (M4Bench), a flexible HRI platform in which multiple users can direct either a single&#x2014;or multiple&#x2014;simulated robots to perform a multi-goal pick-and-place task. Our software exposes a web-based visual interface, with support for mouse, keyboard, gamepad, eye tracker and electromyograph/electroencephalograph (EMG/EEG) user inputs. It can be further extended using native browser libraries or WebSocket interfaces, allowing researchers to add support for their own devices. We also provide tracking for several HRI metrics, such as task completion and command selection time, enabling quantitative comparisons between different user interfaces and devices. We demonstrate the utility of our benchmark with a user study (n &#x3d; 50) conducted to compare five different input devices, and also compare single-vs. multi-user control. In the pick-and-place task, we found that users performed worse when using the eye tracker &#x2b; EMG device pair, as compared to mouse &#x2b; keyboard or gamepad &#x2b; gamepad, over four quantitative metrics (corrected p <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>&#x3c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.001). Our software is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/arayabrain/m4bench">https://github.com/arayabrain/m4bench</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>shared autonomy</kwd>
<kwd>human-robot interaction</kwd>
<kwd>multi-agent</kwd>
<kwd>multimodal</kwd>
<kwd>benchmark</kwd>
</kwd-group>
<funding-group>
<award-group id="gs1">
<funding-source id="sp1">
<institution-wrap>
<institution>Moonshot Research and Development Program</institution>
<institution-id institution-id-type="doi" vocab="open-funder-registry" vocab-identifier="10.13039/open_funder_registry">10.13039/501100020963</institution-id>
</institution-wrap>
</funding-source>
</award-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by JST, Moonshot R&#x26;D Grant Number JPMJMS2012.</funding-statement>
</funding-group>
<counts>
<fig-count count="10"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="62"/>
<page-count count="16"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Robotic Control Systems</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Most human-robot interaction (HRI) research focuses on real robots and specific use-cases, but this can make reproducibility and comparisons between approaches difficult. In contrast, the artificial intelligence community places emphasis on benchmarks in order to track progress in algorithmic development. For instance, many continuous control algorithms are first tested on benchmark tasks in MuJoCo (<xref ref-type="bibr" rid="B52">Todorov et al., 2012</xref>), and later become deployed on real robots.</p>
<p>Driven by this ethos, we developed a multi-agent (<xref ref-type="bibr" rid="B10">Dahiya et al., 2023</xref>), multimodal (<xref ref-type="bibr" rid="B48">Su et al., 2023</xref>) HRI benchmark in order to study the interaction between multiple users and multiple robots (<xref ref-type="fig" rid="F1">Figure 1</xref>), as well as the usability of different input devices, in a shared autonomy paradigm. Whilst we have designed the overall structure of M4Bench to be modular and extensible, its architecture is particularly well-suited for investigating human-robot collaboration (HRC), in which humans and robots work together closely in a shared environment to achieve common goals through mutual interaction and coordination (<xref ref-type="bibr" rid="B2">Ajoudani et al., 2018</xref>; <xref ref-type="bibr" rid="B53">Villani et al., 2018</xref>; <xref ref-type="bibr" rid="B31">Nikolaidis et al., 2017</xref>), and particularly in situations involving shared control and physical manipulation. While other robot types&#x2014;such as quadrupeds or drones&#x2014;are also explored in HRC research, M4Bench currently focuses on robotic arm manipulators, given their widespread adoption and utility in accomplishing collaborative tasks that involve physical interaction.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Usage of M4Bench. <bold>(a)</bold> Multiple users can join the same session for multi-robot control through our web-based server-client. <bold>(b)</bold> M4Bench supports controlling up to 16 robots simultaneously, with info, diagnostics and experimental controls available in the panel on the right.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g001.tif">
<alt-text content-type="machine-generated">Two individuals are seated at a desk, each using a computer to control a user interface displaying multiple robots, with sixteen robots displayed on the screen. The interface shows active connections for two users, along with status buttons and indicators for connected devices.</alt-text>
</graphic>
</fig>
<p>Our goal was to make a flexible benchmark that scales across many dimensions: it supports multiple users, multiple robots, multiple goals (for each robot), and multiple input devices (M4Bench; <xref ref-type="table" rid="T1">Table 1</xref>). Unlike prior benchmarks that perform subsets of these comparisons (<xref ref-type="bibr" rid="B44">Saren et al., 2024</xref>), or that focus primarily on multi-robot systems (<xref ref-type="bibr" rid="B38">Puig et al., 2020</xref>; <xref ref-type="bibr" rid="B60">Zhang et al., 2023</xref>; <xref ref-type="bibr" rid="B29">Mandi et al., 2024</xref>; <xref ref-type="bibr" rid="B13">Esterwood and Robert Jr, 2023</xref>), human-robot coordination (<xref ref-type="bibr" rid="B60">Zhang et al., 2023</xref>; <xref ref-type="bibr" rid="B51">Thumm et al., 2024</xref>; <xref ref-type="bibr" rid="B29">Mandi et al., 2024</xref>), or planning and control of robots or embodied agents via natural language (<xref ref-type="bibr" rid="B29">Mandi et al., 2024</xref>; <xref ref-type="bibr" rid="B9">Chang et al., 2024</xref>), our benchmark enables investigation of the usability, scalability and ease-of-use of different input modalities under multi-user and multi-robot configurations, in a controlled and reproducible setting.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of our benchmark with existing HRC benchmarks (<xref ref-type="sec" rid="s2-2">Section 2.2</xref>) across four axes: multi-user (more than one user), multi-robot (more than one robot), multi-goal (tasks completed by choosing among discrete options; variable setting allows experimenter to change available sub-tasks), and multi-device (more than one input device).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Benchmark</th>
<th align="center">Multi-user</th>
<th align="center">Multi-robot</th>
<th align="center">Multi-goal</th>
<th align="center">Multi-device</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Watch-and-Help (<xref ref-type="bibr" rid="B38">Puig et al., 2020</xref>)</td>
<td align="center">F</td>
<td align="center">F</td>
<td align="center">F</td>
<td align="center">-</td>
</tr>
<tr>
<td align="left">Co-ELA (<xref ref-type="bibr" rid="B60">Zhang et al., 2023</xref>)</td>
<td align="center">F</td>
<td align="center">-</td>
<td align="center">F</td>
<td align="center">F</td>
</tr>
<tr>
<td align="left">The Warehouse Robot Interaction Sim (<xref ref-type="bibr" rid="B13">Esterwood and Robert Jr, 2023</xref>)</td>
<td align="center">F</td>
<td align="center">V</td>
<td align="center">F</td>
<td align="center">F</td>
</tr>
<tr>
<td align="left">Human-Robot Gym (<xref ref-type="bibr" rid="B51">Thumm et al., 2024</xref>)</td>
<td align="center">F</td>
<td align="center">F</td>
<td align="center">F</td>
<td align="center">F</td>
</tr>
<tr>
<td align="left">RoCo (<xref ref-type="bibr" rid="B29">Mandi et al., 2024</xref>)</td>
<td align="center">F</td>
<td align="center">V</td>
<td align="center">V</td>
<td align="center">-</td>
</tr>
<tr>
<td align="left">PARTNR (<xref ref-type="bibr" rid="B9">Chang et al., 2024</xref>)</td>
<td align="center">F</td>
<td align="center">F</td>
<td align="center">V</td>
<td align="center">F</td>
</tr>
<tr>
<td align="left">M4Bench (Ours)</td>
<td align="center">V</td>
<td align="center">V</td>
<td align="center">F</td>
<td align="center">V</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For each axis, we specify the adequate setting among the following three possibilities: F means that the benchmark supports this factor in a fixed configuration; V means that the benchmark supports varying this factor; and - indicates that the modality is either not available or not applicable.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Our implementation addresses several under-explored practical challenges in HRI studies. Firstly, supporting multiple simultaneous human users and robots to be controlled in a shared control loop requires robust session management. Secondly, we built a modular input device interface that not only integrates conventional inputs (e.g., keyboard, mouse), but also biosignal-based devices (e.g., eye trackers, wearable electrodes), thereby allowing researchers to systematically evaluate usability and cognitive load across control modalities. In the current setup for M4Bench, the task (pick-and-place) and robot controllers (inverse kinematics) were deliberately picked to be relatively simple, which allows us to achieve better consistency in quantifying HRI (<xref ref-type="bibr" rid="B61">Zimmerman et al., 2022</xref>). Finally, while various factors may be linked in other benchmarks, M4Bench allows independent and controlled variation across number of users, number of robots, number of goals, and input device combinations, making it possible to isolate key variables in shared autonomy studies.</p>
<p>We also conducted a user study (n &#x3d; 50) to show the utility of our benchmark. In the user study we were able to test hypotheses over two different settings: a comparison over input devices, and a comparison of single-vs. multi-user control. We found significant differences between input devices, and largely no difference between differing numbers of users<xref ref-type="fn" rid="n2">
<sup>1</sup>
</xref>. Our M4Bench software, available at <ext-link ext-link-type="uri" xlink:href="https://github.com/arayabrain/m4bench">https://github.com/arayabrain/m4bench</ext-link>, was designed to be extended, and we hope it will be of use to the HRI community.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<sec id="s2-1">
<label>2.1</label>
<title>Multi-agent HRI</title>
<p>Although most HRI studies involve a single human and single robot, HRI research has evolved to accommodate complex team dynamics that can include multiple users, robots, or both. A system&#x2019;s team composition can be optimized to best suit the unique needs of the task environment at hand.</p>
<p>Multi-user, single-robot collaborative systems have been popular in coordinating search and rescue operations, where the robot is teleoperated in environments too hazardous for human users. In such cases, the collaborating users typically take on different roles. For example, one user may be operating virtual hands while the other monitors robot feedback (<xref ref-type="bibr" rid="B49">Szczurek et al., 2023</xref>).</p>
<p>Alternatively, the team could be composed of a single user and multiple robots. When multiple robots are involved in a system, it is important to consider whether they are homogeneous or heterogeneous. Teams with homogenous robots have been used in the context of human-swarm interaction (HSI) (<xref ref-type="bibr" rid="B14">Gale et al., 2018</xref>). Outside of HSI, operators have controlled homogenous robots to complete industrial workplace tasks in mixed reality (<xref ref-type="bibr" rid="B25">Kennel-Maushart et al., 2023</xref>). Homogeneous frameworks have also conceptually studied to help with the identification of hazardous sources in turbulent environments (<xref ref-type="bibr" rid="B39">Ristic et al., 2017</xref>). Heterogeneous systems have been similarly used in search and rescue, where a single user commands a heterogeneous team of robots based on their capabilities (<xref ref-type="bibr" rid="B27">Liu et al., 2015</xref>). Regardless of the robot team&#x2019;s composition, the human generally takes on the supervisor role that assigns tasks to the multiple robots. However, such systems can put excessive mental workload on a user (<xref ref-type="bibr" rid="B35">Podevijn et al., 2016</xref>). Several solutions have been proposed to combat user mental fatigue, including simplifying the interaction task when the mental load detected is deemed unsustainable for the user (<xref ref-type="bibr" rid="B54">Villani et al., 2020</xref>; <xref ref-type="bibr" rid="B42">Rosenfeld et al., 2017</xref>).</p>
<p>Finally, several teams have attempted to build multi-user multi-robot systems. Although many potential applications are still being evaluated, there have been several notable studies that have provided information on the important factors that contribute to developing such systems. Multi-robot systems have also been designed for social applications. In the classroom, robots have been used as teaching aids, such as helping children learn handwriting (<xref ref-type="bibr" rid="B24">Hood et al., 2015</xref>), or helping middle school students learn about atmospheric science (<xref ref-type="bibr" rid="B33">&#xd6;zg&#xfc;r et al., 2017</xref>). Industrial applications of multi-robot systems have also been explored, such as in automotive manufacturing, where robots contribute to increased efficiency and safety on the production line (<xref ref-type="bibr" rid="B56">Wang et al., 2023</xref>; <xref ref-type="bibr" rid="B25">Kennel-Maushart et al., 2023</xref>).</p>
<p>As HRI systems become increasingly complex and deployed in a wide range of applications, comparing systems has become increasingly difficult. To effectively evaluate how systems compare, we must agree on standardised metrics.</p>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>HRI metrics and benchmarks</title>
<p>Given the diversity of aspects involved in a human-robot interaction, defining metrics that can fully capture every aspect is a complex task. Even more challenging is to define metrics that are generalizable across different studies. Indeed, such metrics would be expected to fit various experimental setups, regardless of the task, the type of robot, the number of users, or the control device employed.</p>
<p>In the early years of HRI, researchers already used a variety of application-specific metrics that were often not directly comparable (<xref ref-type="bibr" rid="B46">Steinfeld et al., 2006</xref>). This was mainly due to the interdisciplinary nature of HRI, which created an inherently decentralized research paradigm (<xref ref-type="bibr" rid="B61">Zimmerman et al., 2022</xref>). This fragmentation hindered the development of unified frameworks and slowed progress in the field. The milestone work of the DARPA/NSF Interdisciplinary Study on Human-Robot Interaction (<xref ref-type="bibr" rid="B41">Rogers and Murphy, 2002</xref>) identified the critical need for standardized metrics in HRI, which <xref ref-type="bibr" rid="B46">Steinfeld et al. (2006)</xref> built upon by introduced a comprehensive set of metrics for HRI, offering structured guidelines for evaluating various aspects of HRI. While much progress has been made since then, HRI metrics still remain an active research area.</p>
<p>Metrics in HRI have adopted a specific configuration across the community, typically categorized based on which aspects of the interaction they measure or evaluate. A survey from 2013 identified forty-two distinct metrics, with seven measuring the human, six measuring the robot, and twenty nine measuring the overeall system (<xref ref-type="bibr" rid="B30">Murphy and Schreckenghost, 2013</xref>).</p>
<p>Metrics can be both explicit qualitative subjective evaluations or implicit quantitative measures. There are five primary methods of evaluation used for human studies in HRI: (1) self-assessments, (2) interviews, (3) behavioural measures, (4) psychophysiology measures, and (5) task performance metrics. As reported in (<xref ref-type="bibr" rid="B6">Bethel and Murphy, 2010</xref>) it seems essential to use three or more methods of evaluation to establish study validity. The use of a single method of measurement is not sufficient to accurately interpret the responses of participants to a robot with which they are interacting. Using more than one way ensures a comprehensive study with reliable and accurate results that can be validated.</p>
<p>Self-assessments are a primary evaluation method in HRI studies, where participants provide direct feedback on their interaction experiences, perceptions, and overall satisfaction with the robot or system. Therefore, the HRI community is increasingly adopting standardized tools such as the NASA task load index (TLX) (<xref ref-type="bibr" rid="B22">Hart, 1988</xref>) for workload assessment and the system usability scale (<xref ref-type="bibr" rid="B7">Brooke, 1996</xref>) for evaluating usability.</p>
<p>On the other hand, performance metrics are based on different aspects of the robotics system, such as accuracy, speed, reliability, robustness, adaptability, scalability, usability, safety, and cost. Depending on the type, domain, and purpose of the robotics system, some metrics may be more relevant than others (<xref ref-type="bibr" rid="B43">Russo, 2022</xref>). Across the most commonly used performance metrics in HRI, we identified several key measures: task completion time; error rate; success rate; efficiency; task accuracy; and interaction effort among others (<xref ref-type="bibr" rid="B46">Steinfeld et al., 2006</xref>; <xref ref-type="bibr" rid="B32">Olsen and Goodrich, 2003</xref>; <xref ref-type="bibr" rid="B23">Hoffman, 2019</xref>).</p>
<p>Given our goal of developing a system capable of adapting to different tasks, robots, users, and control interfaces, we selected the following metrics for implementation: task completion time, command selection time, and error rate, combined with the NASA TLX as a standardised tool for workload assessment. A detailed description of these can be found in <xref ref-type="sec" rid="s3-5">Section 3.5</xref>.</p>
<p>While these metrics can be adapted to specific studies conducted through our platform, they remain primarily suited for comparisons within similar studies and configurations. This underscores the importance of further exploring standardized metrics in HRI. To tackle this challenge, HRI benchmarks play a crucial role in providing a structured framework for testing and evaluation, ensuring consistency and comparability across different subfields and task groups.</p>
<p>Establishing benchmarks that encompass the diverse range of HRI contexts remains a challenge. Nevertheless, several efforts have been made by the research community to unify testing standards within specific categories. One significant advancement has been the adoption of simulated environments for HRI benchmarking. One of the first notable platforms for multi-agent interactions in realistic environments was VirtualHome (<xref ref-type="bibr" rid="B37">Puig et al., 2018</xref>), designed to simulate rich home settings where agents interact with objects and each other. The authors later introduced a benchmark alongside this platform, with a structured evaluation protocol assessing AI agents on success rate, speed-up, and cumulative reward to test generalization and collaboration (<xref ref-type="bibr" rid="B38">Puig et al., 2020</xref>).</p>
<p>Similarly, The Warehouse Robot Interaction Sim is an open-source immersive platform that provides a flexible environment for evaluating cooperative human&#x2013;robot interaction tasks. It features real-time simulation, customizable task scenarios, and adaptive robot behaviours, allowing for in-depth analysis of interaction dynamics and task modifications as needed (<xref ref-type="bibr" rid="B13">Esterwood and Robert Jr, 2023</xref>). Another noteworthy initiative is Human-Robot Gym (<xref ref-type="bibr" rid="B51">Thumm et al., 2024</xref>), which offers HRC benchmarks with diverse collaborative tasks, supports multiple robot systems, and facilitates comprehensive evaluation through predefined tasks and reproducible baselines. Similarly, <xref ref-type="bibr" rid="B29">Mandi et al. (2024)</xref> introduced RoCo, a benchmark with tasks geared toward evaluating the ability of large-language models (LLMs) to control and coordinate robot arms, with the possibility of having a human directly interacting with a robot arm in the real-world while communicating via natural language. To the best of our knowledge, PARTNR (<xref ref-type="bibr" rid="B9">Chang et al., 2024</xref>) represents the most comprehensive benchmarking framework currently available. It integrates multiple evaluation methodologies, supports a wide range of collaborative tasks, and offers the most extensive set of standardized HRI assessments, making it a significant reference point in the field.</p>
<p>However, as shown in <xref ref-type="table" rid="T1">Table 1</xref>, unlike PARTNR and other existing benchmarks, our M4Bench introduces major flexibility across multiple axes, allowing variable configurations for multi-user, multi-robot, and multi-device testing.</p>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>Multimodal HRI</title>
<p>Multimodal HRIs have commonly been implemented in settings with industrial robots, assistive mobile robots, robotic exoskeletons, or robotic prosthetics (<xref ref-type="bibr" rid="B48">Su et al., 2023</xref>). Given that humans naturally communicate through multiple modalities, using multiple input or output devices simultaneously can improve system usability, particularly for users with limited motor control. In elderly users, fusing multiple input modalities has been found to significantly increase human gesture recognition performance (<xref ref-type="bibr" rid="B40">Rodomagoulakis et al., 2016</xref>). Multimodal systems have also been found to benefit hemiplegic users, who showed enhanced engagement and improved movement prediction when combining biological signals like electromyographs (EMG) and electroencephalographs (EEG) during rehabilitation (<xref ref-type="bibr" rid="B19">Gui et al., 2017</xref>).</p>
<p>Early works explored integrating visual and audio input to make intuitive HRI systems (<xref ref-type="bibr" rid="B16">Goodrich and Schultz, 2008</xref>). Since then, the effectiveness of diverse combinations of input modalities has been tested including voice and facial expression (<xref ref-type="bibr" rid="B3">Alonso-Martin et al., 2013</xref>), speech and gesture (<xref ref-type="bibr" rid="B40">Rodomagoulakis et al., 2016</xref>; <xref ref-type="bibr" rid="B47">Strazdas et al., 2022</xref>), and facial expressions with EEG signals (<xref ref-type="bibr" rid="B50">Tan et al., 2021</xref>). Input modalities have been more recently extended to include haptic feedback and physiological sensing (<xref ref-type="bibr" rid="B58">Wang T. et al., 2024</xref>; <xref ref-type="bibr" rid="B12">D&#x2019;Attanasio et al., 2024</xref>). With recent developments in LLMs, LLM-based robotic systems are showing promise in HRI by demonstrating their ability to adapt to multi-modal inputs when determining appropriate assistive actions (<xref ref-type="bibr" rid="B57">Wang C. et al., 2024</xref>; <xref ref-type="bibr" rid="B62">Zu et al., 2024</xref>).</p>
<p>Several studies have compared different input devices for HRI. Some examples include: PS3 gamepad versus PC keyboard (<xref ref-type="bibr" rid="B1">Adamides et al., 2017</xref>), mobile robot control with an app versus gamepad (<xref ref-type="bibr" rid="B28">Mallan et al., 2017</xref>), and robotic navigation with a keypad versus a Nintendo Wii controller (<xref ref-type="bibr" rid="B20">Guo and Sharlin, 2008</xref>). All studies reported significant differences between devices and highlight the importance of selecting appropriate input methods for optimized HRI performance.</p>
<p>While existing HRI benchmarks (<xref ref-type="table" rid="T1">Table 1</xref>) typically focus on agent coordination, language grounding, or simulated avatar control, M4Bench was designed to foreground the human&#x2013;robot interactions themselves&#x2014;specifically in how they scale across different users, robots, goals and input modalities. This introduces several design and engineering challenges, including synchronized multi-user interactions and hardware abstraction for non-traditional control devices. M4Bench offers a foundation for systematically studying shared autonomy across these dimensions, as highlighted in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
</sec>
</sec>
<sec sec-type="materials|methods" id="s3">
<label>3</label>
<title>Materials and methods</title>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> provides an overview of our software, which consists of a web server (running robot simulators), a web interface that receives inputs and displays the robot(s), and, optionally, additional processes to translate inputs from devices such as eye trackers or EMG/EEG. The front-end uses standard HTML, CSS and JS, and is compatible with major browsers (Edge, Safari, Chrome and Firefox). The back-end uses Python and is compatible with Windows, OS X, and Linux. These software architecture choices were made to maximise compatibility and ease of extending the benchmark.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Software diagram. The brain-robot interface (BRI) web application server provides an endpoint to access the user interface, and additionally runs the underlying robot simulator, listening for user commands, and executing them. Users access the interface through a browser, which contains camera feeds from the robot simulator and experiment controls. The browser receives and processes user inputs either through native browser events (e.g., for mouse, keyboard, or gamepad) or dedicated processing modules (e.g., for an eye tracker or EMG/EEG electrodes). Device inputs map to <italic>robot selection</italic> and/or <italic>command selection</italic>.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g002.tif">
<alt-text content-type="machine-generated">Diagram illustrating a system architecture. On the left, &#x22;User&#x22; interacts with &#x22;Devices&#x22; like Mouse, Keyboard, Gamepad, Pupil Core Eye Tracker, and EMG/EEG, which connect to a &#x22;Browser&#x22; in the center. The browser features a Graphical User Interface, Robot Selection, Command Selection, Pupil Core API, and EMG/EEG Decoder. The browser communicates with the &#x22;BRI WebApp Server&#x22; on the right. The server contains a Command Interpreter, Robot Index &#x2b; Command, Robot Policy, and Robot Simulator. Arrows indicate the flow of information between components.</alt-text>
</graphic>
</fig>
<p>We built our web server using the FastAPI framework<xref ref-type="fn" rid="n3">
<sup>2</sup>
</xref>. The web server responds to various HTTP endpoints, runs a multiprocess environment runner, manages WebRTC streams (delivering images from the simulators), and handles user session management. We built the front-end using Bootstrap<xref ref-type="fn" rid="n4">
<sup>3</sup>
</xref> to structure and manage the user interface. As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, the web server, web interface and device processes use WebSockets to communicate real-time information, such as device outputs. A notable exception is the use of WebRTC for streaming video, as it is better suited for such use-cases, and can use peer-to-peer technology to send the images directly from the simulator to the interface.</p>
<p>The environment runner runs one robot simulator process per robot, allowing us to achieve real-time interaction even when scaling up the number of robots to control. We built a pick-and-place environment for our benchmark in RoboHive (<xref ref-type="bibr" rid="B26">Kumar et al., 2023</xref>), a robot learning framework that uses MuJoCo (<xref ref-type="bibr" rid="B52">Todorov et al., 2012</xref>) as its underlying robot simulator. A primary benefit of RoboHive is that it abstracts robot control policies for both real robots and simulated robots to have the same structure, making it easier to develop in a simulator and deploy in the real world: with the right configuration sets, deploying a controller tested in simulation on the real robot is simply a matter of changing an environment flag in RoboHive. Similarly, simulated sensors such as cameras can be replaced by their real-world counterparts by editing the environment configuration files. The other necessity for transferring planning-based manipulation controllers to the real world is object detection, which can be achieved either through ArUco markers (<xref ref-type="bibr" rid="B15">Garrido-Jurado et al., 2014</xref>) or other machine learning/computer vision methods (<xref ref-type="bibr" rid="B4">Bai et al., 2020</xref>).</p>
<p>Our benchmark is set up to allow users to control 1, 4, or 16 robots simultaneously. This allows us to display robots in a regular grid, which simplifies the layout for users. The different numbers of robots allow us to test how human-robot interaction scales across a range of robot numbers, with the single robot scenario also providing a simplified setting with which different input devices can be tested. Simultaneously, multiple users can join an experiment to control robots together by joining from a web browser. It is even possible to allow remote participation, if the web server is made accessible publicly.</p>
<sec id="s3-1">
<label>3.1</label>
<title>Robot task and control</title>
<p>We constructed a simple pick-and-place task for HRI experiments, as our purpose is to test and quantify the interaction between humans and robots, and not the performance of robots at fulfilling complicated tasks. For the task, a robot arm&#x2014;a 7 DoF Franka Panda&#x2014;is placed on the centre of a large square table, with different groups of colored blocks to its sides and front, with bins for each group of blocks placed at the edge of the table. When the robot is instructed to pick a block of the specified color, it will begin picking the specified type of block and placing them in the corresponding bin one by one. We use four groups of two blocks each, which we found provided a good trade-off between goal diversity (number of groups) and robot execution time (amount of time spent picking and placing blocks).</p>
<p>In our task setting, the robot will ignore any other commands until all of the specified blocks are placed. Based on user feedback in early experiments, we added a LED indicator around the base of each robot which lights up when it is active, and remains off when it can be controlled again. As demonstrated in prior work (<xref ref-type="bibr" rid="B5">Baraka et al., 2016</xref>; <xref ref-type="bibr" rid="B36">P&#xf6;rtner et al., 2018</xref>), light indicators are cheap and effective tools for HRI.</p>
<p>The robots are controlled through a simple inverse kinematics motion planner with hard-coded waypoints to place the end-effector above a block, reach down and grasp it, and move it above the bin before opening the gripper. Once the path is planned, the trajectory is executed as fast as possible whilst respecting joint velocity limits. Although this planner does not guarantee 100% task success, in practice we never observed a single failure. However, in order to ensure that experiments can always be completed, if the planner were to fail our software will still count it as a success for the user.</p>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>User interface</title>
<p>When first accessing the user interface via a web browser, the user is directed to a registration page (<xref ref-type="fig" rid="F3">Figure 3a</xref>), where demographic information is collected. After registering, the user is directed to the main menu (<xref ref-type="fig" rid="F3">Figure 3b</xref>). The user can select different input device combinations, and proceed to either data collection or task execution with differing numbers of robots.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>User registration and main menu. <bold>(a)</bold> In user registration, we collect a user name, age, gender, and handedness. <bold>(b)</bold> In the main menu, the user can select between different input device options, participate in data collection, or join a task execution session with either 1, 4, or 16 robots.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g003.tif">
<alt-text content-type="machine-generated">User interface containing two main sections. The first section, labeled &#x22;User registration,&#x22; includes fields for username, age, gender, and handedness with options to clear or register. The second section, labeled &#x22;Main menu,&#x22; allows device selection including Robot and Goal settings, featuring options like Mouse, Gamepad, Eye Tracker, Keyboard, and EEG/EMG. It also presents four mode selection options: Data Collection, Single Robot, Multi Robot with 4 Arms, and Multi Robot with 16 Arms, each depicted with illustrative images of robotic setups. Language options are indicated on the top left of each section.</alt-text>
</graphic>
</fig>
<p>When the user enters a task execution session, they are presented with a view of the robot(s), status information, and experiment controls (<xref ref-type="fig" rid="F4">Figure 4</xref>). Task execution sessions (number of robots) are shared across users, so if multiple users join the same session before it starts, they can jointly control the robot(s). If a task execution session has been started and another user tries to join, they will be blocked until the session has finished. Data collection sessions are not shared, so multiple users can collect data simultaneously.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Task execution interface (4 robots). The centre contains camera views of the robots and their workspaces. The plots in the top-right display predictions over the goals, and also serves as indicators once a goal is selected. From top to bottom, the right panel contains: connected users; device connection status; start and reset experiment buttons; status info; and a collapsible debug log. If the eye tracker device is selected, AprilTags (<xref ref-type="bibr" rid="B55">Wang and Olson, 2016</xref>) are displayed for calibrating the eye tracker&#x2019;s position with respect to the screen.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g004.tif">
<alt-text content-type="machine-generated">Four panels show a robotic arm in a room with colored blocks on a table. The blocks are red, green, and blue. On the right, a user interface displays devices connected, with buttons for starting, stopping, and resetting.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-3">
<label>3.3</label>
<title>Task execution and devices</title>
<p>During task execution, robot selection is performed by mapping a device to the cursor, and moving the cursor onto the camera view(s) (2D continuous control). For this we have implemented support for a mouse, gamepad (joystick), and a Pupil Core eye tracker. Goal (color) selection is performed by mapping a device to the four colors (4D discrete control). For this we have implemented support for a keyboard, gamepad (buttons), and g.tec EMG/EEG devices. Further devices can be added using either native browser libraries or WebSockets.</p>
<p>To prevent users having to recall the color-to-goal associations in our user study (<xref ref-type="sec" rid="s4">Section 4</xref>), we pasted colors on the keyboard keys (1&#x2013;4), and put a diagram of the EMG mapping (<xref ref-type="fig" rid="F5">Figure 5</xref>) on the wall in front of the participants. The colored gamepad buttons could directly be mapped to the goals.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>EMG mapping between muscle contractions and goal colors provided to participants during task execution.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g005.tif">
<alt-text content-type="machine-generated">Diagram of a human outline surrounded by four numbered colored squares with arrows indicating a sequence. Red square one is on the left, green square two is at the bottom left, blue square three is at the bottom right, and yellow square four is on the right. Arrows create a circular flow from one to four.</alt-text>
</graphic>
</fig>
<p>Mouse and keyboard control is achieved using native browser events, and gamepad control is implemented using the native web Gamepad API.</p>
<p>For eye tracker control, we use the Pupil Core API<xref ref-type="fn" rid="n5">
<sup>4</sup>
</xref>, which sends x-y coordinates of the user&#x2019;s fixation (as well as associated confidence values) over a ZeroMQ socket<xref ref-type="fn" rid="n6">
<sup>5</sup>
</xref> to a custom device process. In our preliminary tests and pilot studies, we used the raw Pupil Core API eye tracker mappings to the screen surface, which resulted in erratic cursor movement. Based on iterative testing and users&#x2019; feedback of eye tracking stabilisation methods, we settled on averaging the last eight gaze samples with a confidence <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0.75</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, we send the smoothed values over WebSockets to the browser to control the cursor position.</p>
<p>For EMG/EEG control, we use the g.HIsys Simulink toolbox<xref ref-type="fn" rid="n7">
<sup>6</sup>
</xref> for acquiring and filtering data from g.tec devices, and stream the filtered data using Lab Streaming Layer<xref ref-type="fn" rid="n8">
<sup>7</sup>
</xref> to a custom device process. The device process can record the data, and if given a trained classifier, outputs a predicted goal, as well as a probability distribution over the goals.</p>
</sec>
<sec id="s3-4">
<label>3.4</label>
<title>Data collection</title>
<p>We implemented a data collection mode that presents a randomised sequence of cues (corresponding to the different goals) to the user, and allows us to collect user input data (e.g., EMG signals) for training classification models. The duration of the cues, rest periods, number of trials, and other parameters can be set by the experimenter via the user interface (<xref ref-type="fig" rid="F6">Figure 6</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Data collection interface. A view of the robot is presented at the centre of the screen, with countdowns and cues overlaid during data collection. The experimenter can set the initial countdown, cue duration, rest periods, and the number of blocks (sets of goals).</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g006.tif">
<alt-text content-type="machine-generated">Robotic arm on a wooden table with colored blocks surrounding it. The table has four colored zones: red, yellow, green, and blue. A control panel on the right displays user, device connections, and settings for an experiment, including countdown and cue durations, block number, and estimated duration. Start, stop, and menu buttons are visible.</alt-text>
</graphic>
</fig>
<p>In practice, we collect EMG data and train simple channel-wise threshold-based binary classifiers for each user. To map between EMG signals and the four goals (colors) in the task, we place pairs of electrodes at four sites (the left and right forearms and calves). The users are instructed on the correspondence between sites and goals, i.e., the left forearm maps to red cube, the left calf to the green cube, right calf to the blue cube, and finally the right forearm to the yellow cube. The user is then presented with a video demonstration of the data collection flow during which they practice moving the limb that corresponds to the sequentially displayed color cues until they are comfortable with their performance. The next step is to perform the actual data collection for model calibration as follows: first, a countdown of 2,000 ms is triggered, followed by a random color cue (red, green, blue or yellow) displayed for 2,500 ms during which the users contracts the appropriate limb. An inter-trial rest period of 500 ms is then introduced before proceeding to the next cue. An inter-block rest period of 1,000 ms is introduced every trial block, which is formed by four color cues. We settled for data collection with five blocks, although this can be set by the researcher (<italic>Num Blocks</italic> field in <xref ref-type="fig" rid="F6">Figure 6</xref>) on a per-case basis. This resulted in a total of 20 trials collected over approximately 1 min and 45 s, allowing for quick experiment iterations across users.</p>
<p>We save the EMG data, tagged with the associated goals, in HDF5<xref ref-type="fn" rid="n9">
<sup>8</sup>
</xref>, MNE FIF (<xref ref-type="bibr" rid="B18">Gramfort et al., 2013</xref>) and EEGLAB&#x2019;s <italic>.set</italic> (<xref ref-type="bibr" rid="B11">Delorme and Makeig, 2004</xref>) formats, allowing it to be read easily by several libraries. For each site/EMG channel we train a support vector machine on the maximum signal amplitude (over a 2,500 milliseconds window) to maximise the classification accuracy. With EMG the signal&#x2019;s amplitude increases as the user contracts their muscles, and thus it is straightforward to achieve high accuracies with this simple classifier. The trained classifiers are saved and deployed for the users in the following experiments.</p>
<p>We use Hydra (<xref ref-type="bibr" rid="B59">Yadan, 2019</xref>) and scikit-learn (<xref ref-type="bibr" rid="B34">Pedregosa et al., 2011</xref>) to automatically construct classifier pipelines from YAML files, allowing quick experimentation over different preprocessing steps and model types. For example, the aforementioned threshold classifier is specified as follows:<list list-type="simple">
<list-item>
<p>
<monospace>feature_extractor:</monospace>
</p>
<list list-type="simple">
<list-item>
<p>
<monospace>_target_: feature_extraction.MaximumAmplitude</monospace>
</p>
</list-item>
</list>
</list-item>
<list-item>
<p>
<monospace>vectorizer:</monospace>
</p>
<list list-type="simple">
<list-item>
<p>
<monospace>_target_: mne.decoding.Vectorizer</monospace>
</p>
</list-item>
</list>
</list-item>
<list-item>
<p>
<monospace>classifier:</monospace>
</p>
<list list-type="simple">
<list-item>
<p>
<monospace>_target_: sklearn.svm.SVC</monospace>
</p>
</list-item>
<list-item>
<p>
<monospace>kernel: &#x2019;poly&#x2019;</monospace>
</p>
</list-item>
<list-item>
<p>
<monospace>C: 1</monospace>
</p>
</list-item>
<list-item>
<p>
<monospace>probability: True</monospace>
</p>
</list-item>
</list>
</list-item>
</list>
</p>
<p>Where the pipeline is constructed from composing each item in order, with <monospace>_target_</monospace> specifying a class to instantiate, and other properties specifying the instantiation arguments.</p>
</sec>
<sec id="s3-5">
<label>3.5</label>
<title>Metrics</title>
<p>We selected a set of performance and user experience metrics that capture both task efficiency and cognitive workload. As discussed in <xref ref-type="sec" rid="s2-2">Section 2.2</xref>, identifying metrics that are both meaningful and generalizable across HRC scenarios is a known challenge. Based on a review of commonly used metrics in the literature, we selected those that offer good adaptability across task types and experimental setups.</p>
<p>We evaluate system performance based on the following three quantitative metrics:<list list-type="bullet">
<list-item>
<p>Task Completion Time: the time from the start of the task to its successful completion. This is the most commonly used metric in HRC, as it provides a direct measure of how efficiently the human-robot team completes a given task, and offers a straightforward indicator of overall system performance. In scenarios comparing different input devices for controlling the same system, a shorter task completion time would naturally suggest a more efficient input device. Likewise, when comparing single-user and multi-user collaboration, improved coordination and division of labor in the multi-user setting would be expected to reduce the overall time required. Therefore, the lower the task completion time, the more effective the interface or interaction strategy.</p>
</list-item>
<list-item>
<p>Command Selection Time: represents the time needed for the user to issue a valid command. In the single-robot scenario, it refers to the time the robot waits for a valid input. In the multi-robot case, it captures the time between selecting a robot and confirming a valid goal. This metric is crucial for evaluating the interaction process, as it reflects how quickly users can communicate their intent. It is particularly important when comparing different input modalities or control devices. A lower command selection time indicates that users can issue commands more rapidly, suggesting that the device or interface allows for efficient and fluid interaction. Therefore, systems that minimise this time are generally more intuitive and effective for user control. The metric most similar to ours in definition is the one presented in <xref ref-type="bibr" rid="B45">Shukla et al. (2017)</xref>, where it is referred to as interaction effort or interaction time. Several studies have used similar terms, though definitions vary widely across the literature. We chose to use the term command selection time to avoid confusion about which aspect of the interaction this metric actually measures.</p>
</list-item>
<list-item>
<p>Error Rate: the proportion of invalid goal selection commands sent; the command is invalid if the goal (color choice) is already completed. The entire set of valid/invalid commands are stored, so that summary statistics can be applied afterwards. This metric is essential to assess the accuracy and reliability of the human-robot interaction, as it reflects how often users attempt actions that cannot be executed, highlighting potential issues in user understanding, interface design, or system feedback. A system that enables users to make fewer errors is, by definition, more effective and better designed, as it supports more accurate and reliable interactions.</p>
</list-item>
</list>
</p>
<p>This set of quantitative metrics provides a balanced framework to evaluate key aspects of HRC. By analysing relative differences in values of these metrics, researchers can explore how different factors such as input devices, number of users, or system modifications impact overall performance and interaction quality. This approach also enables the assessment of improvements resulting from changes in system components, such as biosignal classifiers, supporting a systematic and data-driven refinement of HRI systems.</p>
<p>In addition to quantitative performance metrics, evaluating the usability and user experience of the system is essential in human-robot interaction, where task efficiency alone does not fully capture the quality of collaboration. To this end, we selected the NASA TLX as our subjective workload assessment tool. Widely adopted in HRI studies, NASA TLX is a validated and reliable metric that captures users&#x2019; perceived cognitive and physical demands during interaction. Its multidimensional structure makes it particularly suitable for complex, interactive scenarios, such as those involving shared control between humans and robots. We created a webpage for the NASA TLX questionnaire, which users are directed to after completing an experiment (<xref ref-type="fig" rid="F7">Figure 7</xref>). The questionnaire measures the user&#x2019;s perceived workload over six items&#x2014;Mental Demand, Physical Demand, Temporal Demand, Own Performance, Effort, and Frustration Level&#x2014;using a 21-level Likert scale (normalised from 0&#x2013;100, with lower values being better). The individual scores can then be averaged to calculate the overall task load index. Although it is possible to weight the items separately, we stick to the unweighted, &#x201c;raw TLX&#x201d; form (<xref ref-type="bibr" rid="B21">Hart, 2006</xref>), which provides less biased results (<xref ref-type="bibr" rid="B8">Bustamante and Spain, 2008</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>NASA TLX questionnaire. Users are presented with the six items, explanations of each item, and a slider from &#x201c;Very Low&#x201d; to &#x201c;Very High&#x201d; agreement.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g007.tif">
<alt-text content-type="machine-generated">Survey form with six sections: Mental Demand, Physical Demand, Temporal Demand, Performance, Effort, and Mental. Each section features a slider ranging from Very Low to Very High, with current positions at Very Low or Perfect. Clear and Submit buttons are at the bottom.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-6">
<label>3.6</label>
<title>Logging</title>
<p>We save user (demographic) data, experiment metrics, and questionnaire results in a hierarchical folder structure that resembles BIDS (<xref ref-type="bibr" rid="B17">Gorgolewski et al., 2016</xref>), but which separates experiment-specific and user-specific data into different folders. All of this data is stored in JSON format to be both human- and machine-readable.</p>
</sec>
<sec id="s3-7">
<label>3.7</label>
<title>Localisation</title>
<p>In order to enable users with different native languages to use our software comfortably, we implemented a language toggle (currently supporting English and Japanese), available on the user registration and main interface. This allows the experimenter/user to dynamically set the language on the user interface during experiments in order to accommodate users from different linguistic backgrounds. This feature is particularly important for collecting questionnaire data, so that questions can be conveyed in the user&#x2019;s native language. The localisation code allows additional languages to be added by adding translations for relevant text to a single localisation file, where text for the interface is extracted from.</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<label>4</label>
<title>Results</title>
<sec id="s4-1">
<label>4.1</label>
<title>Hypotheses</title>
<p>In order to demonstrate the capabilities of our benchmark, we designed and ran a user study to test several hypotheses, under two settings (<xref ref-type="table" rid="T2">Table 2</xref>). In the first setting, we investigated differences between pairs of input devices for robot and goal selection: mouse &#x2b; keyboard; gamepad &#x2b; gamepad; and eye tracker &#x2b; EMG. In this setting, we have a single user controlling four robots, with the three different devices pairs. In the second setting, we investigated the differences between single- and multi-user control. In this setting, we have either a single user or two users control 16 robots, using the mouse &#x2b; keyboard device combination. For both settings, the hypotheses we test are:<list list-type="simple">
<list-item>
<p>H1: There are differences in the mean task completion time.</p>
</list-item>
<list-item>
<p>H2: There are differences in the mean Command Selection Time.</p>
</list-item>
<list-item>
<p>H3: There are differences in the mean error rate.</p>
</list-item>
<list-item>
<p>H4: There are differences in the mean overall task load index.</p>
</list-item>
</list>
</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>User study experimental settings. We either varied the input devices (experiment 1) or number of users (experiment 2).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Experiment</th>
<th align="center">&#x23; Users</th>
<th align="center">&#x23; Robots</th>
<th align="center">Devices</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">4</td>
<td align="center">mouse &#x2b; keyboard vs. gamepad &#x2b; gamepad vs. eye tracker &#x2b; EMG</td>
</tr>
<tr>
<td align="center">2</td>
<td align="center">1 vs. 2</td>
<td align="center">16</td>
<td align="center">mouse &#x2b; keyboard</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The most effective device combination and number of users would ideally lead to a shorter task completion time (H1), a faster command selection time (H2), a lower error rate (H3), and a lower level of perceived workload (H4). Such a combination can provide a practical upper bound on performance, serving as a baseline for evaluating alternative interfaces. If another combination delivers command selection times comparable to this baseline, it can be considered functionally competitive. Lower error rates may also suggest benefits beyond accessibility, such as easier use or reduced cognitive effort.</p>
<p>We note that we provide these analyses as suggestions for system evaluation methods, and do not claim that one setup in necessarily superior to another. For example, our modular benchmark also allows for identifying setups tailored to individual users, accommodating diverse user needs and preferences.</p>
</sec>
<sec id="s4-2">
<label>4.2</label>
<title>User study</title>
<p>We recruited 50 volunteers for our user study (18 female, four left-handed, with an age distribution of 28.1 <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 7.2 years), forming 25 pairs for single-vs. multi-user control. At the beginning of the study, each user was briefed on the experiments, and asked to sign a consent form. If they consented, we proceeded with the set of experiments. <xref ref-type="fig" rid="F8">Figure 8</xref> shows the flow of the user study for pairs of users. Upon completion of the study, users were given a gift card. Our study was given ethical approval by the Shiba Palace Clinic Ethics Review Committee.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Overview of the user studies. We investigated three different combinations of devices, i.e., &#x201c;Mouse and Keyboard&#x201d;, &#x201c;Gamepad and Gamepad&#x201d;, and &#x201c;Eye Tracker and EMG electrodes&#x201d;. As illustrated in the pipeline, the first user performs the device comparison study with four robots, then starts the single-vs. multi-user comparison study, with the second user joining in the middle. The second user then performs the rest of their experiments in reverse order.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g008.tif">
<alt-text content-type="machine-generated">Overview of a study on device modalities. Top section lists devices: mouse, keyboard, gamepad, eye-tracker, EMG. Pipeline shows users interacting with devices; modalities include single-user and multi-user. Study A compares devices for a single user with four robots. Study B compares single-user versus multi-user with sixteen robots.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4-3">
<label>4.3</label>
<title>Input device comparison</title>
<p>The overall results for the input device comparison setting are reported in <xref ref-type="table" rid="T3">Table 3</xref> and the detailed NASA TLX results are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>. We checked that the data for our hypotheses was approximately Gaussian-distributed, and then ran a repeated measures ANOVA test (<inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.05). This test revealed significant effects of device type on Task Completion Time [F(2, 98) &#x3d; 62.81, <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>], Command Selection Time [F(2, 98) &#x3d; 26.74, <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>], Error Rate [F(2, 98) &#x3d; 54.27, <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>], and Overall Workload [F(2, 98) &#x3d; 34.66, <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>]. We then conducted <italic>post hoc</italic> Tukey HSD tests to examine pairwise differences. The eye tracker &#x2b; EMG device pair performed significantly worse across all metrics. No significant differences were found between the mouse &#x2b; keyboard and gamepad &#x2b; gamepad input devices. Bonferroni correction was applied to each <italic>p</italic>-value from the pairwise comparisons to account for multiple comparisons (correction factor &#x3d; 3 per metric, capped at 1.0):</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>HRI metrics for different combinations of devices for robot and goal selection, with a single user controlling four robots. Average &#xb1; 1 standard deviation reported over 50 participants.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Robot-goal selection</th>
<th align="center">Task completion time (s)</th>
<th align="center">Command selection time (s)</th>
<th align="center">Error rate</th>
<th align="center">Overall workload</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Mouse &#x2b; Keyboard</td>
<td align="center">105.6 &#xb1; 4.2</td>
<td align="center">0.423 &#xb1; 0.443</td>
<td align="center">0.005 &#xb1; 0.029</td>
<td align="center">14.7 &#xb1; 13.5</td>
</tr>
<tr>
<td align="left">Gamepad &#x2b; Gamepad</td>
<td align="center">106.9 &#xb1; 5.8</td>
<td align="center">0.596 &#xb1; 1.155</td>
<td align="center">0.014 &#xb1; 0.041</td>
<td align="center">17.4 &#xb1; 15.3</td>
</tr>
<tr>
<td align="left">Eye Tracker &#x2b; EMG</td>
<td align="center">132.7 &#xb1; 22.5</td>
<td align="center">0.846 &#xb1; 1.284</td>
<td align="center">0.292 &#xb1; 0.263</td>
<td align="center">40.8 &#xb1; 20.1</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>NASA TLX scores for different combinations of devices for robot and goal selection, with a single user controlling four robots. Average &#xb1; 1 standard deviation reported over 50 participants.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g009.tif">
<alt-text content-type="machine-generated">Bar chart comparing workload across six categories: Mental Demand, Physical Demand, Temporal Demand, Own Performance, Effort, and Frustration Level. Three input methods are shown: Mouse and Keyboard in blue, Gamepad in orange, and Eye Tracker and EMG in green. Eye Tracker and EMG generally show higher workload levels in most categories. Error bars indicate variability.</alt-text>
</graphic>
</fig>
<p>Task completion time</p>
<p>
<list list-type="bullet">
<list-item>
<p>Eye tracker &#x2b; EMG vs. gamepad &#x2b; gamepad: M &#x3d; &#x2212;25.71; T &#x3d; &#x2212;9.32, <bold>corrected</bold> <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Eye tracker &#x2b; EMG vs. mouse &#x2b; keyboard: M &#x3d; &#x2212;27.08; T &#x3d; &#x2212;9.82, <bold>corrected</bold> <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Mouse &#x2b; keyboard vs. gamepad &#x2b; gamepad: M &#x3d; 1.37; T &#x3d; 0.50, corrected p &#x3d; 1.0</p>
</list-item>
</list>
</p>
<p>Command selection time</p>
<p>
<list list-type="bullet">
<list-item>
<p>Eye tracker &#x2b; EMG vs. gamepad &#x2b; gamepad: M &#x3d; &#x2212;0.35; T &#x3d; &#x2212;4.98, <bold>corrected</bold> <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Eye tracker &#x2b; EMG vs. mouse &#x2b; keyboard: M &#x3d; &#x2212;0.52; T &#x3d; &#x2212;7.47, <bold>corrected</bold> <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Mouse &#x2b; keyboard vs. gamepad &#x2b; gamepad: M &#x3d; 0.17; T &#x3d; 2.49, corrected p &#x3d; 0.1109</p>
</list-item>
</list>
</p>
<p>Error rate</p>
<p>
<list list-type="bullet">
<list-item>
<p>Eye tracker &#x2b; EMG vs. gamepad &#x2b; gamepad: M &#x3d; &#x2212;0.28; T &#x3d; &#x2212;8.90, <bold>corrected</bold> <inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Eye tracker &#x2b; EMG vs. mouse &#x2b; keyboard: M &#x3d; &#x2212;0.29; T &#x3d; &#x2212;9.18, <bold>corrected</bold> <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Mouse &#x2b; keyboard vs. gamepad &#x2b; gamepad: M &#x3d; 0.01; T &#x3d; 0.28, corrected p &#x3d; 1.0</p>
</list-item>
</list>
</p>
<p>Overall workload</p>
<p>
<list list-type="bullet">
<list-item>
<p>Eye tracker &#x2b; EMG vs. gamepad &#x2b; gamepad: M &#x3d; &#x2212;23.38; T &#x3d; &#x2212;7.00, <bold>corrected</bold> <inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Eye tracker &#x2b; EMG vs. mouse &#x2b; keyboard: M &#x3d; &#x2212;26.12; T &#x3d; &#x2212;7.82, <bold>corrected</bold> <inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Mouse &#x2b; keyboard vs. gamepad &#x2b; gamepad: M &#x3d; 2.73; T &#x3d; 0.82, corrected p &#x3d; 1.0</p>
</list-item>
</list>
</p>
<p>The individual NASA TLX results complement this finding, as the eye tracker &#x2b; EMG device pair was deemed more demanding to use across all items. Direct behavioural analysis from observing particpants also supports this, with the setup time and concentration required to perform the task with the eye tracker &#x2b; EMG combination increasing the workload on the users.</p>
</sec>
<sec id="s4-4">
<label>4.4</label>
<title>Number of users comparison</title>
<p>The overall results for the number of users comparison setting are reported in <xref ref-type="table" rid="T4">Table 4</xref> and the detailed NASA TLX results are shown in <xref ref-type="fig" rid="F10">Figure 10</xref>. We checked that the data for our hypotheses was approximately Gaussian-distributed, and then ran a paired t-test (<inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.05). The resulting <italic>p</italic>-values were Bonferroni-corrected to account for multiple comparisons (correction factor &#x3d; 2). No significant differences were found in Task Completion Time, Error Rate, or Overall Workload. There were no noticeable differences in the individual NASA TLX results. After correction, we found a significant difference in Command Selection Time between one and two users controlling 16 robots [t(49) &#x3d; &#x2212;3.71, <bold>corrected</bold> <inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi mathvariant="bold">p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn mathvariant="bold">0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>], with two users taking longer to interact. However, because the multi-user runs always included a novice user by design, this introduces a confound, which we were able to confirm using a t-test on the difference in the mean command selection time between the first user and second user in a pair. Whilst we therefore cannot draw conclusions on H2, we leave this detail to illustrate the utility of M4Bench&#x2019;s detailed metric logging.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>HRI metrics for single-vs. multi-user robot control, with users controlling 16 robots. Average &#xb1; 1 standard deviation reported over 50 participants.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">&#x23; Users</th>
<th align="center">Task completion time (s)</th>
<th align="center">Command selection time (s)</th>
<th align="center">Error rate</th>
<th align="center">Overall workload</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">One</td>
<td align="center">122.9 &#xb1; 11.7</td>
<td align="center">0.237 &#xb1; 0.282</td>
<td align="center">0.007 &#xb1; 0.033</td>
<td align="center">22.5 &#xb1; 18.8</td>
</tr>
<tr>
<td align="left">Two</td>
<td align="center">122.6 &#xb1; 9.5</td>
<td align="center">0.319 &#xb1; 0.594</td>
<td align="center">0.001 &#xb1; 0.004</td>
<td align="center">19.9 &#xb1; 15.7</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>NASA TLX scores for single-vs. multi-user robot control, with users controlling 16 robots. Average &#xb1; 1 standard deviation reported over 50 participants.</p>
</caption>
<graphic xlink:href="frobt-12-1528754-g010.tif">
<alt-text content-type="machine-generated">Bar chart comparing workload across six categories for single users and two users. Categories include mental, physical, temporal demand, own performance, effort, and frustration level. Single users are shown in red, and two users in purple, with error bars representing variability.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4-5">
<label>4.5</label>
<title>General observations</title>
<p>Beyond each device&#x2019;s intrinsic usability, prior experience with the devices played an important role in performance across tasks. Nearly all users were highly accustomed to keyboard and mouse setups, having used them regularly. This familiarity enabled efficient performance, with only minor differences across age groups. Conversely, elderly users were unfamiliar with gamepads, and were slightly less proficient with them. Finally, the eye tracker and EMG combination was entirely new to participants in our user study, and the brief practice sessions we conducted failed to make up for the extensive experience gap; anecdotally, the authors themselves are able to achieve similar results using all device pairs.</p>
<p>We also observed that the Performance item in the NASA TLX questionnaire was interpreted differently depending on the users. Some users rated themselves purely on whether the task was completed, whilst others rated themselves based on how long they took. This is of course a common issue with qualitative questionnaires.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<label>5</label>
<title>Discussion</title>
<p>The flexibility of our software makes it a suitable testbed for investigating HRC with different input devices, as a precursor to more challenging tasks. For instance, this platform would allow iterating on decoding algorithms for EMG/EEG, before deploying them to real-world settings. Whilst we believe our benchmark is useful by itself, it also has greater potential for trialling novel HRI approaches that can then be ported to different scenarios.</p>
<p>While M4Bench is currently centred on HRC&#x2014;with an emphasis on physical interaction, shared control, and collaborative manipulation&#x2014;its modular and extensible architecture offers a foundation for broader applications. There are numerous improvements and additional functionalities that could be implemented in the system. For instance, adding support for new devices will enhance the platform&#x2019;s flexibility, allowing it to accommodate a broader range of interfaces and enable more versatile HRI studies. Expanding customization options for device-specific parameters would make the system even more adaptable, especially for complex devices like EEG/EMG. Such devices offer a wide variety of configurations&#x2014;from adjusting the number of classes to choosing paradigms and customizing training. Providing options to fine-tune these details would give researchers greater control, allowing them to optimize the system for diverse experimental needs and usage contexts.</p>
<p>Moreover, enhancing the platform with additional metrics would significantly improve its adaptability and relevance across diverse research contexts. This enhancement could involve adding both more system-calculated metrics and standardized qualitative measures. To further support customization, the platform could also allow researchers to pre-select the metrics most relevant to their specific study needs.</p>
<p>Although the platform currently calculates the metrics automatically, the analysis of the results is performed externally. A valuable enhancement would be to integrate automated analysis directly within the system. This could include the ability to compare different experimental conditions, generate detailed performance reports, and provide real-time insights, offering researchers an efficient and seamless way to evaluate their data without needing additional tools. This could improve the overall research workflow and allow for a more comprehensive understanding of the outcomes directly within the platform.</p>
<p>The system could include built-in basic tasks as a starting point, offering ready-made configurations for standard experimental scenarios. These basic tasks could also serve as templates, which researchers could customize further to suit their specific experimental goals.</p>
<p>All these features would enable researchers to tailor the M4Bench platform in detail to meet their specific objectives, making our system versatile, robust, and adaptable to a wide range of research needs and environments.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because of privacy concerns. Requests to access the datasets should be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Shiba Palace Clinic Ethics Review Committee. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>AY: Investigation, Methodology, Software, Writing &#x2013; original draft. RD: Investigation, Methodology, Software, Writing &#x2013; original draft. MD: Data curation, Formal Analysis, Investigation, Methodology, Software, Writing &#x2013; original draft. SS: Methodology, Software, Writing &#x2013; original draft. HD: Formal Analysis, Investigation, Software, Writing &#x2013; original draft. KA: Conceptualization, Formal Analysis, Project administration, Writing &#x2013; original draft.</p>
</sec>
<ack>
<title>Acknowledgements</title>
<p>The authors would like to thank Masaru Kuwabara and Shogo Akiyama for their help with the preparation for user studies. We used <ext-link ext-link-type="uri" xlink:href="https://www.jikken-baito.com">https://www.jikken-baito.com</ext-link> for recruitment of participants.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>Authors AY, RD, MD, SS, HD, and KA were employed by Araya Inc.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2019445/overview">Jing Yao</ext-link>, Chinese Academy of Sciences (CAS), China</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2212942/overview">Bin-Bin Hu</ext-link>, University of Groningen, Netherlands</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3118237/overview">Mohammad Hassan Farhadi</ext-link>, University of Rhode Island, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3163326/overview">Alessia De Nobile</ext-link>, Roma Tre University, Italy</p>
</fn>
</fn-group>
<fn-group>
<fn id="n2">
<label>1</label>
<p>Our investigations explained away the one significant difference found as an experimental artifact.</p>
</fn>
<fn id="n3">
<label>2</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://fastapi.tiangolo.com/">https://fastapi.tiangolo.com/</ext-link>
</p>
</fn>
<fn id="n4">
<label>3</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://getbootstrap.com/">https://getbootstrap.com/</ext-link>
</p>
</fn>
<fn id="n5">
<label>4</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://docs.pupil-labs.com/core/developer/network-api/">https://docs.pupil-labs.com/core/developer/network-api/</ext-link>
</p>
</fn>
<fn id="n6">
<label>5</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://zeromq.org/">https://zeromq.org/</ext-link>
</p>
</fn>
<fn id="n7">
<label>6</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.gtec.at/product/g-hisys/">https://www.gtec.at/product/g-hisys/</ext-link>
</p>
</fn>
<fn id="n8">
<label>7</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://github.com/sccn/labstreaminglayer">https://github.com/sccn/labstreaminglayer</ext-link>
</p>
</fn>
<fn id="n9">
<label>8</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.hdfgroup.org/solutions/hdf5/">https://www.hdfgroup.org/solutions/hdf5/</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adamides</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Katsanos</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Parmet</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Christou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Xenos</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hadzilacos</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Hri usability evaluation of interaction modes for a teleoperated agricultural robotic sprayer</article-title>. <source>Appl. Ergon.</source> <volume>62</volume>, <fpage>237</fpage>&#x2013;<lpage>246</lpage>. <pub-id pub-id-type="doi">10.1016/j.apergo.2017.03.008</pub-id>
<pub-id pub-id-type="pmid">28411734</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ajoudani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zanchettin</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Ivaldi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Albu-Sch&#xe4;ffer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kosuge</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Khatib</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Progress and prospects of the human&#x2013;robot collaboration</article-title>. <source>Aut. robots</source> <volume>42</volume>, <fpage>957</fpage>&#x2013;<lpage>975</lpage>. <pub-id pub-id-type="doi">10.1007/s10514-017-9677-2</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alonso-Martin</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Malfaz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sequeira</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gorostiza</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Salichs</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A multimodal emotion detection system during human&#x2013;robot interaction</article-title>. <source>Sensors</source> <volume>13</volume>, <fpage>15549</fpage>&#x2013;<lpage>15581</lpage>. <pub-id pub-id-type="doi">10.3390/s131115549</pub-id>
<pub-id pub-id-type="pmid">24240598</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Object detection recognition and robot grasping based on machine learning: a survey</article-title>. <source>IEEE access</source> <volume>8</volume>, <fpage>181855</fpage>&#x2013;<lpage>181879</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.3028740</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baraka</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Paiva</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Veloso</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Expressive lights for revealing mobile service robot state</article-title>. <source>Robot 2015 Second Iber. Robotics Conf. Adv. Robotics</source> <volume>1</volume>, <fpage>107</fpage>&#x2013;<lpage>119</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-27146-0_9</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bethel</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Review of human studies methods in hri and recommendations</article-title>. <source>Int. J. Soc. Robotics</source> <volume>2</volume>, <fpage>347</fpage>&#x2013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1007/s12369-010-0064-9</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brooke</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Sus-a quick and dirty usability scale</article-title>. <source>Usability Eval. industry</source> <volume>189</volume>, <fpage>4</fpage>&#x2013;<lpage>7</lpage>.</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bustamante</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Spain</surname>
<given-names>R. D.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Measurement invariance of the nasa tlx</article-title>. <source>Proc. Hum. factors ergonomics Soc. Annu. Meet.</source> <volume>52</volume>, <fpage>1522</fpage>&#x2013;<lpage>1526</lpage>. <pub-id pub-id-type="doi">10.1177/154193120805201946</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chhablani</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Clegg</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cote</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Desai</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hlavac</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Partnr: a benchmark for planning and reasoning in embodied multi-agent tasks</article-title>. <source>arXiv Prepr. arXiv:2411.00081</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2402.00081</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Dahiya</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Aroyo</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Dautenhahn</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2023</year>). <source>A survey of multi-agent human-robot interaction systems</source>. <publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name>.</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Delorme</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Makeig</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Eeglab: an open source toolbox for analysis of single-trial eeg dynamics including independent component analysis</article-title>. <source>J. Neurosci. methods</source> <volume>134</volume>, <fpage>9</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1016/j.jneumeth.2003.10.009</pub-id>
<pub-id pub-id-type="pmid">15102499</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>D&#x2019;Attanasio</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alabert</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Francis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Studzinska</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Exploring multimodal interactions with a robot assistant in an assembly task: a human-centered design approach</article-title>,&#x201d; in <source>VISIGRAPP, GRAPP, HUCAPP, IVAPP</source> (<publisher-loc>Set&#x00FA;bal, Portugal</publisher-loc>: <publisher-name>SciTePress</publisher-name>), <volume>1</volume>, <fpage>549</fpage>&#x2013;<lpage>556</lpage>. <pub-id pub-id-type="doi">10.5220/0012570800003660</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Esterwood</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Robert Jr</surname>
<given-names>L. P.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>The warehouse robot interaction sim: an open-source hri research platform</article-title>,&#x201d; in <source>Companion of the 2023 ACM/IEEE international conference on human-robot interaction</source>, <fpage>268</fpage>&#x2013;<lpage>271</lpage>.</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Gale</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Karasinski</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hillenius</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Playbook for uas: ux of goal-oriented planning and execution</article-title>,&#x201d; in <source>Engineering psychology and cognitive ergonomics: 15th international conference, EPCE 2018, held as part of HCI international 2018, Las Vegas, NV, USA, July 15-20, 2018, proceedings 15</source> (<publisher-name>Springer</publisher-name>), <fpage>545</fpage>&#x2013;<lpage>557</lpage>.</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garrido-Jurado</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mu&#xf1;oz-Salinas</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Madrid-Cuevas</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Mar&#xed;n-Jim&#xe9;nez</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Automatic generation and detection of highly reliable fiducial markers under occlusion</article-title>. <source>Pattern Recognit.</source> <volume>47</volume>, <fpage>2280</fpage>&#x2013;<lpage>2292</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2014.01.005</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodrich</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Schultz</surname>
<given-names>A. C.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Human&#x2013;robot interaction: a survey</article-title>. <source>Found. Trends&#xae; Human&#x2013;Computer Interact.</source> <volume>1</volume>, <fpage>203</fpage>&#x2013;<lpage>275</lpage>. <pub-id pub-id-type="doi">10.1561/1100000005</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gorgolewski</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Auer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Calhoun</surname>
<given-names>V. D.</given-names>
</name>
<name>
<surname>Craddock</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Duff</surname>
<given-names>E. P.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>The brain imaging data structure, a format for organizing and describing outputs of neuroimaging experiments</article-title>. <source>Sci. data</source> <volume>3</volume>, <fpage>160044</fpage>&#x2013;<lpage>160049</lpage>. <pub-id pub-id-type="doi">10.1038/sdata.2016.44</pub-id>
<pub-id pub-id-type="pmid">27326542</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Luessi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Larson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Engemann</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Strohmeier</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Brodbeck</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Meg and eeg data analysis with mne-python</article-title>. <source>Front. Neurosci.</source> <volume>267</volume>, <fpage>267</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2013.00267</pub-id>
<pub-id pub-id-type="pmid">24431986</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gui</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Toward multimodal human&#x2013;robot interaction to enhance active participation of users in gait rehabilitation</article-title>. <source>IEEE Trans. Neural Syst. Rehabilitation Eng.</source> <volume>25</volume>, <fpage>2054</fpage>&#x2013;<lpage>2066</lpage>. <pub-id pub-id-type="doi">10.1109/TNSRE.2017.2703586</pub-id>
<pub-id pub-id-type="pmid">28504943</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sharlin</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Exploring the use of tangible user interfaces for human-robot interaction: a comparative study</article-title>. <source>Proc. SIGCHI Conf. Hum. Factors Comput. Syst.</source>, <fpage>121</fpage>&#x2013;<lpage>130</lpage>. <pub-id pub-id-type="doi">10.1145/1357054.1357076</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hart</surname>
<given-names>S. G.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Nasa-task load index (Nasa-tlx); 20 years later</article-title>. <source>Proc. Hum. factors ergonomics Soc. Annu. Meet.</source> <volume>50</volume>, <fpage>904</fpage>&#x2013;<lpage>908</lpage>. <pub-id pub-id-type="doi">10.1177/154193120605000909</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hart</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Staveland</surname>
<given-names>L. E.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>Development of nasa-tlx (task load index): results of empirical and theoretical research</article-title>. <source>Hum. Ment. workload/Elsevier</source>, <fpage>139</fpage>&#x2013;<lpage>183</lpage>. <pub-id pub-id-type="doi">10.1016/s0166-4115(08)62386-9</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hoffman</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Evaluating fluency in human&#x2013;robot collaboration</article-title>. <source>IEEE Trans. Human-Machine Syst.</source> <volume>49</volume>, <fpage>209</fpage>&#x2013;<lpage>218</lpage>. <pub-id pub-id-type="doi">10.1109/thms.2019.2904558</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Hood</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lemaignan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dillenbourg</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>When children teach a robot to write: an autonomous teachable humanoid which uses simulated handwriting</article-title>,&#x201d; in <source>Proceedings of the tenth annual ACM/IEEE international conference on human-robot interaction</source>, <fpage>83</fpage>&#x2013;<lpage>90</lpage>.</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Kennel-Maushart</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Poranne</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Coros</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Interacting with multi-robot systems <italic>via</italic> mixed reality</article-title>,&#x201d; in <source>2023 IEEE international conference on robotics and automation (ICRA)</source> (<publisher-name>IEEE</publisher-name>), <fpage>11633</fpage>&#x2013;<lpage>11639</lpage>.</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Shah</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Moens</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Caggiano</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Vakil</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>RoboHive: a unified framework for robot learning</article-title>. <source>arXiv</source>, <fpage>06828</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2302.06828</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ficocelli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nejat</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>A supervisory control method for multi-robot task allocation in urban search and rescue</article-title>,&#x201d; in <source>2015 IEEE international symposium on safety, security, and rescue robotics (SSRR)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mallan</surname>
<given-names>V. S.</given-names>
</name>
<name>
<surname>Gopi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Muir</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bhavani</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Comparative empirical usability assessment of two hri input devices for a mobile robot</article-title>,&#x201d; in <source>2017 4th international conference on signal processing, computing and control (ISPCC)</source> (<publisher-name>IEEE</publisher-name>), <fpage>331</fpage>&#x2013;<lpage>337</lpage>.</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mandi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Roco: dialectic multi-robot collaboration with large language models</article-title>,&#x201d; in <source>2024 IEEE international conference on robotics and automation (ICRA)</source>, <fpage>286</fpage>&#x2013;<lpage>299</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610855</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Murphy</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Schreckenghost</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Survey of metrics for human-robot interaction</article-title>,&#x201d; in <source>2013 8th ACM/IEEE international conference on human-robot interaction (HRI)</source> (<publisher-name>IEEE</publisher-name>), <fpage>197</fpage>&#x2013;<lpage>198</lpage>.</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Nikolaidis</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nath</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Procaccia</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Srinivasa</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Game-theoretic modeling of human adaptation in human-robot collaboration</article-title>,&#x201d; in <source>Proceedings of the 2017 ACM/IEEE international conference on human-robot interaction</source>, <fpage>323</fpage>&#x2013;<lpage>331</lpage>.</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Olsen</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Goodrich</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Metrics for evaluating human-robot interactions</article-title>. <source>Proc. PERMIS (Citeseer)</source> <volume>2003</volume>, <fpage>4</fpage>.</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>&#xd6;zg&#xfc;r</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Johal</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Mondada</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dillenbourg</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Windfield: learning wind meteorology with handheld haptic robots</article-title>,&#x201d; in <source>Proceedings of the 2017 ACM/IEEE international conference on human-robot interaction</source>, <fpage>156</fpage>&#x2013;<lpage>165</lpage>.</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Scikit-learn: machine learning in python</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>. <pub-id pub-id-type="doi">10.5555/1953048.2078195</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Podevijn</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>O&#x2019;grady</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mathews</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gilles</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fantini-Hauwel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dorigo</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Investigating the effect of increasing robot group sizes on the human psychophysiological state in the context of human&#x2013;swarm interaction</article-title>. <source>Swarm Intell.</source> <volume>10</volume>, <fpage>193</fpage>&#x2013;<lpage>210</lpage>. <pub-id pub-id-type="doi">10.1007/s11721-016-0124-3</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>P&#xf6;rtner</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schr&#xf6;der</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rasch</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sprute</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hoffmann</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>K&#xf6;nig</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>The power of color: a study on the effective use of colored light in human-robot interaction</article-title>,&#x201d; in <source>2018 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source> (<publisher-name>IEEE</publisher-name>), <fpage>3395</fpage>&#x2013;<lpage>3402</lpage>.</mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Puig</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ra</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Boben</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fidler</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Virtualhome: simulating household activities <italic>via</italic> programs</article-title>,&#x201d; in <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <fpage>8494</fpage>&#x2013;<lpage>8502</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2010.09890</pub-id>
</mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Puig</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J. B.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Watch-and-help: a challenge for social perception and human-ai collaboration</article-title>. <source>arXiv Prepr. arXiv:2010.09890</source>.</mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ristic</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Angley</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Moran</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Palmer</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Autonomous multi-robot search for a hazardous source in a turbulent environment</article-title>. <source>Sensors</source> <volume>17</volume>, <fpage>918</fpage>. <pub-id pub-id-type="doi">10.3390/s17040918</pub-id>
<pub-id pub-id-type="pmid">28430120</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Rodomagoulakis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kardaris</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pitsikalis</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Mavroudi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Katsamanis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tsiami</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Multimodal human action recognition in assistive human-robot interaction</article-title>,&#x201d; in <source>2016 IEEE international conference on acoustics, speech and signal processing (ICASSP)</source> (<publisher-name>IEEE</publisher-name>), <fpage>2702</fpage>&#x2013;<lpage>2706</lpage>.</mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Rogers</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Human-robot interaction: final report for darpa/nsf study on human-robot interaction</article-title>. <comment>Technical Report</comment>. <publisher-loc>San Luis Obispo, CA</publisher-loc>: <publisher-name>California Polytechnic State University</publisher-name>.</mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rosenfeld</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Agmon</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Maksimov</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Kraus</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Intelligent agent supporting human&#x2013;multi-robot team collaboration</article-title>. <source>Artif. Intell.</source> <volume>252</volume>, <fpage>211</fpage>&#x2013;<lpage>231</lpage>. <pub-id pub-id-type="doi">10.1016/j.artint.2017.08.005</pub-id>
</mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Russo</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Measuring performance: metrics for manipulator design, control, and optimization</article-title>. <source>Robotics</source> <volume>12</volume> (<issue>4</issue>), <fpage>4</fpage>. <pub-id pub-id-type="doi">10.3390/robotics12010004</pub-id>
</mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mukhopadhyay</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ghose</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Biswas</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Comparing alternative modalities in the context of multimodal human-robot interaction</article-title>. <source>JMUI</source> <volume>18</volume>, <fpage>69</fpage>&#x2013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1007/s12193-023-00421-w</pub-id>
</mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Shukla</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Erkent</surname>
<given-names>&#xd6;.</given-names>
</name>
<name>
<surname>Piater</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Proactive, incremental learning of gesture-action associations for human-robot collaboration</article-title>,&#x201d; in <source>2017 26th IEEE international symposium on robot and human interactive communication (RO-MAN)</source> (<publisher-name>IEEE</publisher-name>), <fpage>346</fpage>&#x2013;<lpage>353</lpage>.</mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Steinfeld</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kaber</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lewis</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Scholtz</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schultz</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). &#x201c;<article-title>Common metrics for human-robot interaction</article-title>,&#x201d; in <source>Proceedings of the 1st ACM SIGCHI/SIGART conference on Human-robot interaction</source>, <fpage>33</fpage>&#x2013;<lpage>40</lpage>.</mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Strazdas</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hintz</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Khalifa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abdelrahman</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Hempel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Al-Hamadi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Robot system assistant (rosa): towards intuitive multi-modal and multi-device human-robot interaction</article-title>. <source>Sensors</source> <volume>22</volume>, <fpage>923</fpage>. <pub-id pub-id-type="doi">10.3390/s22030923</pub-id>
<pub-id pub-id-type="pmid">35161671</pub-id>
</mixed-citation>
</ref>
<ref id="B48">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sandoval</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Laribi</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Recent advancements in multimodal human&#x2013;robot interaction</article-title>. <source>Front. Neurorobotics</source> <volume>17</volume>, <fpage>1084000</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2023.1084000</pub-id>
<pub-id pub-id-type="pmid">37250671</pub-id>
</mixed-citation>
</ref>
<ref id="B49">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Szczurek</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Prades</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Matheson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rodriguez-Nogueira</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Di Castro</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Multimodal multi-user mixed reality human&#x2013;robot interface for remote operations in hazardous environments</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>17305</fpage>&#x2013;<lpage>17333</lpage>. <pub-id pub-id-type="doi">10.1109/access.2023.3245833</pub-id>
</mixed-citation>
</ref>
<ref id="B50">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sol&#xe9;-Casals</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Caiafa</surname>
<given-names>C. F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A multimodal emotion recognition method based on facial expressions and electroencephalography</article-title>. <source>Biomed. Signal Process. Control</source> <volume>70</volume>, <fpage>103029</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2021.103029</pub-id>
</mixed-citation>
</ref>
<ref id="B51">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Thumm</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Trost</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Althoff</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Human-robot gym: benchmarking reinforcement learning in human-robot collaboration</article-title>,&#x201d; in <source>2024 IEEE international conference on robotics and automation (ICRA)</source> (<publisher-name>IEEE</publisher-name>), <fpage>7405</fpage>&#x2013;<lpage>7411</lpage>.</mixed-citation>
</ref>
<ref id="B52">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Todorov</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Erez</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tassa</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2012</year>). <source>MuJoCo: a physics engine for model-based control</source> in <source>IROS</source>.</mixed-citation>
</ref>
<ref id="B53">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Villani</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Pini</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Leali</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Secchi</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Survey on human&#x2013;robot collaboration in industrial settings: safety, intuitive interfaces and applications</article-title>. <source>Mechatronics</source> <volume>55</volume>, <fpage>248</fpage>&#x2013;<lpage>266</lpage>. <pub-id pub-id-type="doi">10.1016/j.mechatronics.2018.02.009</pub-id>
</mixed-citation>
</ref>
<ref id="B54">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Villani</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Capelli</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Secchi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fantuzzi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sabattini</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Humans interacting with multi-robot systems: a natural affect-based approach</article-title>. <source>Aut. Robots</source> <volume>44</volume>, <fpage>601</fpage>&#x2013;<lpage>616</lpage>. <pub-id pub-id-type="doi">10.1007/s10514-019-09889-6</pub-id>
</mixed-citation>
</ref>
<ref id="B55">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Olson</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Apriltag 2: efficient and robust fiducial detection</article-title>,&#x201d; in <source>2016 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source> (<publisher-name>IEEE</publisher-name>), <fpage>4193</fpage>&#x2013;<lpage>4198</lpage>.</mixed-citation>
</ref>
<ref id="B56">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ore</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hauge</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Meijer</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Multi-actor perspectives on human robotic collaboration implementation in the heavy automotive manufacturing industry-a swedish case study</article-title>. <source>Technol. Soc.</source> <volume>72</volume>, <fpage>102165</fpage>. <pub-id pub-id-type="doi">10.1016/j.techsoc.2022.102165</pub-id>
</mixed-citation>
</ref>
<ref id="B57">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hasler</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tanneberg</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ocker</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Joublin</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ceravola</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). &#x201c;<article-title>Lami: large language models for multi-modal human-robot interaction</article-title>,&#x201d; in <source>Extended abstracts of the CHI conference on human factors in computing systems</source>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>.</mixed-citation>
</ref>
<ref id="B58">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>Multimodal human&#x2013;robot interaction for human-centric smart manufacturing: a survey</article-title>. <source>Adv. Intell. Syst.</source> <volume>6</volume>, <fpage>2300359</fpage>. <pub-id pub-id-type="doi">10.1002/aisy.202300359</pub-id>
</mixed-citation>
</ref>
<ref id="B59">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yadan</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Hydra - a framework for elegantly configuring complex applications</article-title>. <source>Github</source>.</mixed-citation>
</ref>
<ref id="B60">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Shan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J. B.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Building cooperative embodied agents modularly with large language models</article-title>. <source>arXiv Prepr. arXiv:2307.02485</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2307.02485</pub-id>
</mixed-citation>
</ref>
<ref id="B61">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zimmerman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bagchi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Marvel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>An analysis of metrics and methods in research from human-robot interaction conferences, 2015-2021</article-title>,&#x201d; in <source>HRI</source>.</mixed-citation>
</ref>
<ref id="B62">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Language and sketching: an llm-driven interactive multimodal multitask robot navigation framework</article-title>,&#x201d; in <source>2024 IEEE international conference on robotics and automation (ICRA)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1019</fpage>&#x2013;<lpage>1025</lpage>.</mixed-citation>
</ref>
</ref-list>
</back>
</article>