<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="brief-report" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1346714</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2024.1346714</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Brief Research Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A pipeline for estimating human attention toward objects with on-board cameras on the iCub humanoid robot</article-title>
<alt-title alt-title-type="left-running-head">Hanifi et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2024.1346714">10.3389/frobt.2024.1346714</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Hanifi</surname>
<given-names>Shiva</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2589844/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Maiettini</surname>
<given-names>Elisa</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1334191/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name>
<surname>Lombardi</surname>
<given-names>Maria</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/445487/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Natale</surname>
<given-names>Lorenzo</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/36032/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>Humanoid Sensing and Perception Group</institution>, <institution>Istituito Italiano di Tecnologia</institution>, <addr-line>Genoa</addr-line>, <country>Italy</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/948419/overview">Tapomayukh Bhattacharjee</ext-link>, Cornell University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2634254/overview">Ziang Liu</ext-link>, Cornell University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2813330/overview">Tom Silver</ext-link>, Princeton University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2743346/overview">Pranav Thakkar</ext-link>, Sibley School of Mechanical and Aerospace Engineering, College of Engineering, Cornell University, Ithaca, United States, in collaboration with reviewer TS</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Shiva Hanifi, <email>shiva.hanifi@iit.it</email>; Maria Lombardi, <email>maria.lombardi1@iit.it</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1346714</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>09</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Hanifi, Maiettini, Lombardi and Natale.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Hanifi, Maiettini, Lombardi and Natale</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>This research report introduces a learning system designed to detect the object that humans are gazing at, using solely visual feedback. By incorporating face detection, human attention prediction, and online object detection, the system enables the robot to perceive and interpret human gaze accurately, thereby facilitating the establishment of joint attention with human partners. Additionally, a novel dataset collected with the humanoid robot iCub is introduced, comprising more than 22,000 images from ten participants gazing at different annotated objects. This dataset serves as a benchmark for human gaze estimation in table-top human&#x2013;robot interaction (HRI) contexts. In this work, we use it to assess the proposed pipeline&#x2019;s performance and examine each component&#x2019;s effectiveness. Furthermore, the developed system is deployed on the iCub and showcases its functionality. The results demonstrate the potential of the proposed approach as a first step to enhancing social awareness and responsiveness in social robotics. This advancement can enhance assistance and support in collaborative scenarios, promoting more efficient human&#x2013;robot collaborations.</p>
</abstract>
<kwd-group>
<kwd>attention</kwd>
<kwd>gaze estimation</kwd>
<kwd>learning architecture</kwd>
<kwd>humanoid robot</kwd>
<kwd>computer vision</kwd>
<kwd>human&#x2013;robot scenario</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Human-Robot Interaction</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Any face-to-face interaction between two people is characterized by a continuous exchange of social signals, such as gaze, gestures, and facial expressions. Such non-verbal communication is possible because interacting individuals can see, perceive, and understand the social information enclosed in cues. In this study, we prioritize eye gaze, a critical social cue, because it plays a pivotal role in many mechanisms of social cognition, for example, joint attention, regulating and monitoring turn-taking, signaling attention, and intention. Neuropsychological evidence highlighted the close relationship between gaze direction and attention, indicating that gaze functions are actively involved and influenced by spatial attention systems (<xref ref-type="bibr" rid="B3">Allison et al., 2000</xref>; <xref ref-type="bibr" rid="B33">Pelphrey et al., 2003</xref>). For example, it is more likely that the gaze is directed toward an object rather than toward empty space.</p>
<p>In this context, a robot&#x2019;s ability to determine what a human is looking at (e.g., an object) has numerous practical implications across various domains. In social robotics, it enhances a robot&#x2019;s social awareness and responsiveness, making interactions more natural and context-appropriate (<xref ref-type="bibr" rid="B5">Babel et al., 2021</xref>; <xref ref-type="bibr" rid="B18">Holman et al., 2021</xref>). This includes recognizing a person&#x2019;s preferences based on their gaze and improving collaboration in settings like industry or home by understanding human attention (<xref ref-type="bibr" rid="B19">Kurylo and Wilson, 2019</xref>).</p>
<p>This research report represents the initial milestone in our ongoing study aiming at interpreting human intent during human&#x2013;robot collaboration. We introduce a novel HRI application utilizing computer vision to enable robots to detect the object a human partner is gazing at. This application sets the baseline for forthcoming advancements in our research. Our proposed system combines an online object detection algorithm (<xref ref-type="bibr" rid="B11">Ceola et al., 2021</xref>; <xref ref-type="bibr" rid="B26">Maiettini et al., 2019a</xref>) with gaze tracking technologies, providing the robot with online information about the objects that capture the human&#x2019;s attention. This integration grants the robot enhanced cognitive ability to perceive and interpret human gaze accurately in its environment. This could be the initial step in enabling the robot to achieve conscious joint attention with the human partner (<xref ref-type="bibr" rid="B13">Chevalier et al., 2020</xref>).</p>
<p>The main contributions are as follows:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We propose a pipeline to detect the target of human attention during an interaction with a robot. This leverages face detection, human attention prediction, and online object detection to detect the object the human focuses on.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We present the <italic>ObjectDetection</italic> dataset collected with the humanoid iCub (<xref ref-type="bibr" rid="B31">Metta et al., 2010</xref>), where 10 participants gaze at different objects placed randomly on a table in front of the robot, including annotations of ground truth gaze target and object bounding boxes.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We perform an experimental analysis of the proposed pipeline to evaluate its effectiveness in the considered HRI setting. We use the collected dataset to do that, and we study the performance of the components of the system.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Finally, we deploy the system on the iCub robot. A video is submitted as <xref ref-type="sec" rid="s13">Supplementary Material</xref>.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>The problem of endowing robots with the capability to comprehend human behavior, particularly the social cue of the gaze, has been studied in the literature. In this regard, the human line of sight, which consists of two main components&#x2014;the head pose and the orientation of the eyes within their sockets (eyegaze) (<xref ref-type="bibr" rid="B37">Wang and Sung, 2002</xref>)&#x2014;offers critical information for predicting human attention and intention. Although extended literature addresses the use of egocentric gaze data from external wearable devices [e.g., head-mounted eye trackers (<xref ref-type="bibr" rid="B1">Admoni and Srinivasa, 2016</xref>) and chest-mounted cameras (<xref ref-type="bibr" rid="B17">Furnari et al., 2017</xref>; <xref ref-type="bibr" rid="B6">Bertasius et al., 2016</xref>)] or using a geometric approach to estimate gaze [where the eyes and pupils need to be clearly visible in the image (<xref ref-type="bibr" rid="B32">Palinko et al., 2015</xref>)], our study upholds a naturalistic HRI setting by avoiding external devices utilizing a third-person view and positioning the human partner at a distance from the robot.</p>
<p>In this context, the gaze problem is addressed following two different strategies: 1) gaze estimation (i.e., estimating the gaze vector or mutual gaze events) and 2) gaze attention prediction (i.e., understanding where the human is visually attending in terms of a saliency map).</p>
<p>Following the <italic>gaze estimation strategy</italic>, <xref ref-type="bibr" rid="B37">Wang and Sung (2002)</xref> employ zoom-in iris imaging to estimate eye gaze from a single eye. They integrate head pose and eye gaze determination for enhanced accuracy. Other works focus on human gaze estimation using a 2D/3D vector. For example, the use of the CNN architecture to estimate the 2D gaze vector is proposed by <xref ref-type="bibr" rid="B4">Athavale et al. (2022)</xref>. This system extracts features from only one eye and is especially useful in real-world conditions where the human face can be partially obscured. In this regard, <xref ref-type="bibr" rid="B16">Fischer et al. (2018)</xref> propose a novel dataset of varied gaze and head pose images in a natural environment, addressing the issue of ground truth annotation by measuring head pose using a motion capture system and eye gaze using mobile eye-tracking glasses. Examples of predicting a 3D gaze vector can be found in <xref ref-type="bibr" rid="B12">Cheng et al. (2020)</xref> and <xref ref-type="bibr" rid="B36">Ververas et al. (2022)</xref>. Specifically, <xref ref-type="bibr" rid="B36">Ververas et al. (2022)</xref> propose an architecture to estimate the vector of the gaze direction from the reconstructed dense 3D eyeball meshes. <xref ref-type="bibr" rid="B12">Cheng et al. (2020)</xref>, instead, propose a combination of a regression and an evaluation network able to exploit the asymmetry between the left and right eye. Additionally, <xref ref-type="bibr" rid="B21">Lombardi et al. (2022a)</xref> propose a learning architecture to detect mutual gaze events. This study underscores the significance of mutual gaze as a vital social cue in face-to-face interactions, indicating the readiness of interacting partners.</p>
<p>The <italic>gaze following</italic> problem was addressed by <xref ref-type="bibr" rid="B34">Recasens et al. (2017)</xref>. A CNN architecture was proposed, taking the RGB frame and a set of neighboring frames from the same video as input and identifying which of the neighboring frames, if any, contain the object being looked at and the coordinates of the human gaze.</p>
<p>Even though gaze estimation and human attention have been extensively studied, few works have integrated human attention with target object prediction. Among these few, <xref ref-type="bibr" rid="B35">Saran et al. (2018)</xref> proposed an approach to predict the human referential gaze, having both the person and object of attention visible in the image. The proposed network contains two pathways: one estimating the head direction and another for salient objects in the scene. Such a network was used as a backbone by <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref>. In the latter, differently from <xref ref-type="bibr" rid="B35">Saran et al. (2018)</xref>, an LSTM-based spatio-temporal model is used to leverage the temporal coherence of video frames to improve gaze direction estimation. However, only the direction of human gaze is predicted by <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref>, while the information about the target object is not provided.</p>
<p>In this report, we adapt the LSTM-based spatio-temporal model from <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> to an HRI setting, specifically a table-top scenario where the robot and human partner are positioned on opposite sides of a table, with the human and objects within the robot&#x2019;s field of view. We fine-tune the model using the proposed <italic>ObjectAttention</italic> dataset, which is annotated with both object bounding boxes and the gazed target object. Additionally, we integrate it with human pose estimation and a face detector to enable real-time processing on the iCub robot. Using human pose estimation alongside an RGB-based face detector rather than an eye-tracking system was motivated by our commitment to have a natural HRI. Furthermore, studies suggested that humans shift their gaze, moving first the head and then the eyeballs in a linear and coordinated way, known as eye-head coordination (<xref ref-type="bibr" rid="B23">Maesako and Koike, 1993</xref>; <xref ref-type="bibr" rid="B29">Melvill Jones et al., 1988</xref>). Such eye-head temporal coordination especially characterizes conscious situations (contrarily, situations in which eyes precede the head movements are processed at an unconscious level) (<xref ref-type="bibr" rid="B15">Doshi and Trivedi, 2009</xref>). Finally, by integrating an online object detection method, we allow the system to predict the class label and location of the gaze target object. Note that, unlike <xref ref-type="bibr" rid="B35">Saran et al. (2018)</xref>, by using <xref ref-type="bibr" rid="B27">Maiettini et al. (2019b)</xref> for object detection, the entire system can be easily adapted to detect novel target objects in only a few seconds. All the mentioned improvements result in an online robotic application that makes the robot capable of inferring where the human partner&#x2019;s attention is targeted while interacting with them.</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<p>The proposed pipeline is made of three pathways (<xref ref-type="fig" rid="F1">Figure 1</xref>): the <italic>Human Attention Estimation</italic> pathway aiming at detecting the attention target of the human, the <italic>Object Detection</italic> pathway that recognizes and localizes the objects in the scene, and the <italic>Attentive Object Detection</italic> pathway that provides the gazed object from the human.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Pipeline of the presented architecture. The <italic>Human Attention Estimation</italic> pathway produces a heatmap of the gaze target. This heatmap and the bounding boxes and labels from the <italic>Object Detection</italic> module are then used by the <italic>Attentive Object Detection</italic> module to predict the specific object that is visually attended by the human.</p>
</caption>
<graphic xlink:href="frobt-11-1346714-g001.tif"/>
</fig>
<sec id="s3-1">
<title>3.1 Human attention estimation</title>
<p>The <italic>Human Attention Estimation</italic> pathway has three distinct modules: 1) Human Pose Estimation, 2) Face Detection, and 3) Visual Target Detection. Having the RGB image as input, the final output of this pathway is the real-time prediction of the human attention target, provided as a heatmap.</p>
<sec id="s3-1-1">
<title>3.1.1 Human pose estimation</title>
<p>We rely on the OpenPose architecture proposed by <xref ref-type="bibr" rid="B9">Cao et al. (2019)</xref>. In brief, OpenPose is a system for multi-human pose estimation that receives as input RGB frames and predicts the location in pixel <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> of 135 anatomical keypoints of each person in the image. It also associates a confidence level <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to each prediction. The choice of <italic>Human Pose Estimation</italic> is motivated by having access to anatomical keypoints, facilitating further applications such as action recognition.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Face detection</title>
<p>We rely on the face recognition presented by <xref ref-type="bibr" rid="B22">Lombardi et al. (2022b)</xref> to detect and extract the human face from the image. Specifically, the face keypoints extracted by the <italic>Human Pose Estimation</italic> module are used as input, while the output is the bounding box of the person&#x2019;s head in front of the robot. Note that <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> assume that the information of the face location is available. That is a strong limitation in applying the method in online robotic applications, preventing it from being used on real robots. In this work, we provide the online input to the <italic>Visual Target Detection</italic> module by using <italic>Human Pose Estimation</italic> together with <italic>Face Detection</italic>, enabling the pipeline to operate on the actual robot.</p>
</sec>
<sec id="s3-1-3">
<title>3.1.3 Visual target detection</title>
<p>This module takes as input the RGB image from the robot camera and the human face bounding box extracted by the <italic>Face Detection</italic> module. It provides as output the heatmap representing the image area that more likely contains the target of human attention. Specifically, this is an image-sized matrix where each cell corresponds to an image pixel. The value of each cell ranges from 0 to 1 (respectively, the lowest and the highest probability to be &#x2013;or to be close to&#x2013; the target of human attention). For this module, we rely on the network presented by <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref>, which is composed of three main parts. The first one is the <italic>Head Conditioning Branch</italic>, which uses the head bounding box encoded into a convolutional feature map (head feature map) together with the information of the location of the human&#x2019;s head in the image to predict a first attention map. The second part is the <italic>Main Scene Branch</italic>, which multiplies the convolutional feature map of the entire image with the attention map and concatenates the result with the previously computed head feature map. The final tensor represents the input for the third and last part, namely, the <italic>Recurrent Attention Prediction Branch</italic>. This first encodes the tensor used as input for a convolutional long short-term memory network, then creates the final attention heatmap by upsampling the latter&#x2019;s output using a decoder. In this work, we fine-tune the network&#x2019;s weights using our dataset, and the resulting model is used for the developed application and the experimental analysis.</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Object detection</title>
<p>The <italic>Object Detection</italic> pathway is characterized by one module that takes the RGB images from the robot&#x2019;s camera as input and outputs the bounding boxes of all the objects of interest present in the scene. For this task, we rely on the online object detection approach presented by <xref ref-type="bibr" rid="B11">Ceola et al. (2021)</xref> and <xref ref-type="bibr" rid="B26">Maiettini et al. (2019a)</xref>. This Mask R-CNN-based system is easily retrainable online, ensuring swift adaptation without compromising performance. We train the online object detection with data acquired using the pipeline described by <xref ref-type="bibr" rid="B25">Maiettini et al. (2017)</xref>.</p>
</sec>
<sec id="s3-3">
<title>3.3 Attentive object detection</title>
<p>The third pathway combines the extracted information from human attention with the objects in the scene to detect the object that is the target of the human gaze. It takes as input the RGB image, the heatmap from the <italic>Visual Target Detection</italic> module, and all the bounding boxes and labels predicted by the <italic>Object Detection</italic> pathway. The output is the attended object bounding box and label.</p>
<p>Initially, the heatmap undergoes thresholding to isolate the region with values surpassing a refined threshold (the hottest part of the heatmap). This process aims to pinpoint the area indicative of human gaze focus within the image. Then, we compute the center of the obtained area and the surrounding bounding box. We use this information to select the object that is the most likely focus of human attention. Precisely, we choose the object that either presents a higher value of intersection over union (IoU) with the bounding box of the hottest part of the heatmap or, if this latter does not intersect any object bounding box, we select the object whose center is the closest to the center of the hottest part.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Dataset</title>
<p>A major contribution of this work is the <italic>ObjectAttention</italic> dataset. It depicts HRIs in a table-top scenario where the human gazes at different objects, and the robot understands the gaze direction and the target object.</p>
<sec id="s4-1">
<title>4.1 Data collection</title>
<p>We recruited 10 participants (four women and six men) with normal or corrected vision (six people wore glasses). Data collection was conducted with the iCub robot (<xref ref-type="bibr" rid="B31">Metta et al., 2010</xref>), and all participants provided written informed consent. To collect the dataset, the iCub was positioned on one side of a table, with a RealSense 415 camera<xref ref-type="fn" rid="fn2">
<sup>1</sup>
</xref>mounted on its head. We placed up to five objects from the YCB dataset (<xref ref-type="bibr" rid="B8">Calli et al., 2015</xref>) on the table in various arrangements. The layout and object mix were different for each participant. The participants were instructed to stand on the other side of the table, facing the robot and looking at the requested object in a natural and spontaneous manner. The frames were recorded using the RealSense 415 camera and the YARP middleware (<xref ref-type="bibr" rid="B30">Metta et al., 2006</xref>).</p>
<p>We collected data in five sessions with each participant, starting with one object in the scene and gradually increasing the number of objects up to five. We performed two trials for each session, keeping the same number of objects but changing the object types and their arrangements on the table. For each session and trial, we collected a 5 s video for each different object, annotating the gazed target object as ground truth.</p>
<p>The resulting dataset consists of 250 videos (22,732 frames) depicting 10 participants in two different trials for each of the five sessions, gazing at the different objects. Additionally, for at least one trial per session, we placed a distracting object (i.e., the <italic>Pringles</italic> object) on the table, at which the participant was not asked to gaze. Details and example frames are reported in the <xref ref-type="sec" rid="s13">Supplementary Material</xref>.</p>
<p>Finally, our motivation to collect a new dataset is that the dataset of <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> contains more conditions in which the gaze was directed toward the upper part of the map (not suitable for a table-top). Our dataset, used to fine-tune the learning model, was collected for scenarios where the human and the robot look at objects placed on a table. <xref ref-type="fig" rid="F2">Figure 2</xref> depicts the density map of the gaze targets for the dataset in <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> (b) and the one we collected (c).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>
<bold>(A)</bold> Selection of sample output frames of the proposed pipeline. The first row depicts the scene image, as well as the head bounding box of the participant detected by the <italic>Face Detection</italic> module, the attention heatmap of the participant, and the bounding box of the hottest area of the heatmap. The second row depicts the related gaze target selections for the frames of the first row. <bold>(B)</bold> Gaze target location density for the dataset used in <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> and <bold>(C)</bold> gaze target location density for the <italic>ObjectAttention</italic> dataset.</p>
</caption>
<graphic xlink:href="frobt-11-1346714-g002.tif"/>
</fig>
</sec>
<sec id="s4-2">
<title>4.2 Data annotation</title>
<p>Each setting requires bounding boxes for the participant&#x2019;s head and the target object. The participants&#x2019; bounding box was extracted using the keypoints estimated by <italic>Openpose</italic> (<xref ref-type="bibr" rid="B10">Cao et al., 2017</xref>) and manually refined to be considered as ground truth. Furthermore, we manually annotated the bounding boxes and classes for all the objects on the table, highlighting the one that is the target of the human&#x2019;s attention. The gaze target point was chosen as the center of the gazed object. The bounding box labeling was done using the <italic>LabelImg</italic>
<xref ref-type="fn" rid="fn3">
<sup>2</sup>
</xref> framework.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Experiments</title>
<sec id="s5-1">
<title>5.1 Model training</title>
<p>Both the <italic>Object Detection</italic> and the <italic>Visual Target Detection</italic> modules were re-trained to better suit the considered conditions.</p>
<sec id="s5-1-1">
<title>5.1.1 Object detection training</title>
<p>We trained the online object detection with data acquired using the pipeline described in <xref ref-type="bibr" rid="B25">Maiettini et al. (2017)</xref>. Specifically, a human teacher showed the objects of interest to the robot, one at a time, holding them in their hand and moving them in front of the robot for approximately 30 s. The information from the robot&#x2019;s depth sensors was used to localize the object and follow it with the robot&#x2019;s gaze. The latter can be segmented, and the corresponding bounding box was automatically assigned through a depth segmentation routine (i.e., the learning object is the closest to the robot&#x2019;s camera) and gathered as ground truth together with the object&#x2019;s label, provided verbally. After each object demonstration, the collected data were used to update the current object detection model.</p>
</sec>
<sec id="s5-1-2">
<title>5.1.2 Visual target detection fine-tuning</title>
<p>To fine-tune the <italic>Visual Target Detection</italic> module, we randomly split the <italic>ObjectAttention</italic> dataset by participants, considering approximately <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mn>70</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the dataset (data from seven participants) as a training set, and the remaining <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mn>30</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as a test set, ensuring no overlap of data between the train and test splits. We fine-tuned the spatio-temporal model of the <italic>Visual Target Detection</italic> module on the training set, performing a warm training re-start with the pre-trained weights provided by the authors <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> and empirically choosing the hyper-parameters as follows: <italic>learning rate</italic> <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <italic>batch size</italic> <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <italic>chunk size</italic> <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <italic>number of epochs</italic> <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. To ensure the statistical relevance of the presented experiments, we repeated the training and evaluation of the model three times with three different splits of the dataset.</p>
</sec>
</sec>
<sec id="s5-2">
<title>5.2 Experimental setup</title>
<p>The performance of the <italic>Visual Target Detection</italic> module is evaluated in terms of the <italic>area under the curve</italic> (<italic>AUC</italic>) and <italic>Distance</italic> metrics. For the AUC, each cell in the spatially discretized image is classified as either the gaze target or not. The ground truth comes from thresholding a Gaussian confidence mask centered at the human annotator&#x2019;s target location. The final heatmap provides the prediction confidence score evaluated at different thresholds in the ROC curve. The <italic>AUC</italic> of this ROC curve is considered. The <italic>Distance</italic> metric is defined as the <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x141;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> distance between the annotated target location and the prediction given by the pixel of the maximum value in the heatmap, with image width and height normalized to 1. The performance for the entire pipeline is measured in terms of the <italic>Accuracy</italic> of the detected gazed objects. For each image, the bounding box of the predicted gazed object is compared with the ground truth: if the gazed object is correctly identified, the prediction is counted as a true positive; otherwise, it is considered a false negative.</p>
</sec>
<sec id="s5-3">
<title>5.3 Visual target detection fine-tuning</title>
<p>First, we analyze the impact of fine-tuning the <italic>Visual Target Detection</italic> module on our dataset. In <xref ref-type="table" rid="T1">Table 1</xref>, we report the performance comparison of the proposed model (row <bold>Fine-tuned model</bold>) with the model presented in <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> (row <bold>Pre-trained model</bold>) in terms of mean and standard deviation over the three dataset splits mentioned above. As can be seen, the fine-tuned model reports better performance on the proposed <italic>ObjectAttention</italic> dataset. Specifically, the predicted hottest point in the heatmap is closer to the true gazed point of <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.04. Note that this is a relevant difference because the <italic>Distance</italic> metric is computed on an image with width and height normalized to 1. This result is also supported by the improvement in the <italic>AUC</italic> of <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. To quantify the distance metric in the task space, we used the depth information and the intrinsic camera parameters to calculate the Euclidean distance between the 3D coordinates of the center of the ground truth bounding box of the gaze target object and the center of the predicted bounding box of the gaze target object. It results in a task space distance of <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mn>0.092</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>0.127</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> meters.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Quantitative evaluation of the <italic>Visual Target Detection</italic> model on the presented <italic>ObjectAttention</italic> dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">AUC (%) <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:mi>&#x2191;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x141;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> distance <inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:mi>&#x2193;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Pre-trained model</td>
<td align="center">87.5 <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.9</td>
<td align="center">0.131 <inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.014</td>
</tr>
<tr>
<td align="center">Fine-tuned model</td>
<td align="center">92.5 <inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1.9</td>
<td align="center">0.089 <inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.014</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The enhanced performance stems from fine-tuning the model with a dataset more aligned with the target scenario (table-top). Nevertheless, because the fine-tuned network has been initialized with the weights presented in <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref>, the final model can predict gaze directions that differ from those considered in the proposed dataset (see the video provided as <xref ref-type="sec" rid="s13">Supplementary Material</xref>).</p>
</sec>
<sec id="s5-4">
<title>5.4 Accuracy evaluation</title>
<p>In order to evaluate the performance of the overall pipeline, we choose one of the models trained on the three different train/test splits and use it in our pipeline. Quantitative results are obtained using the same test set previously employed for evaluating the fine-tuned model, with ground truth provided by bounding boxes and labels of objects on the table and the target gazed object.</p>
<p>First, we analyze the overall accuracy of the pipeline in detecting the gazed target object of the three different participants in the test set. Our experiments indicate a success rate of 79.5% in correctly detecting the target object. This number reflects the integrated performance of the <italic>Visual Target Detection</italic>, <italic>Object Detection</italic>, and <italic>Attentive Object Detection</italic> modules.</p>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> illustrates a selection of sample frames from the output, including attention heatmaps and bounding boxes, highlighting the head of the participant (detected by the <italic>Face Detection</italic> module) and the hottest areas of the heatmap in the frames of the top row while the final gaze object bounding box and label are presented in the bottom row frames (see also in the video in the <xref ref-type="sec" rid="s13">Supplementary Material</xref>).</p>
<p>With the aim to be in line with the current state-of-the-art, we benchmarked a visual language model (VLM) to evaluate the overall accuracy. We choose the open-source LLAVA-1.6 model as the VLM (<xref ref-type="bibr" rid="B20">Liu et al., 2024</xref>), which reports a success rate of 15% in correctly detecting the gazed object. The very poor performance is explained by the fact that a VLM is not targeted to solve a highly specific task like the one reported in this report. More details are reported in the <xref ref-type="sec" rid="s13">Supplementary Material</xref>.</p>
</sec>
<sec id="s5-5">
<title>5.5 Performance analysis</title>
<sec id="s5-5-1">
<title>5.5.1 Per object performance</title>
<p>In <xref ref-type="fig" rid="F3">Figure 3B</xref>, we present the achieved accuracy levels for various target objects. The system demonstrates high performance across most objects, except for the <italic>Bleach</italic> class. This discrepancy arises from challenges in object detection, leading to occasional inaccuracies in locating the <italic>Bleach</italic> object. Such issues may result from disparities between the detector&#x2019;s training conditions and the testing environment, indicating a domain shift. Previous studies have suggested addressing this issue through methods such as integrating autonomous exploration by robots in new domains and employing weakly supervised learning techniques (<xref ref-type="bibr" rid="B27">Maiettini et al., 2019b</xref>; <xref ref-type="bibr" rid="B28">Maiettini et al., 2021</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Accuracy analysis: <bold>(A)</bold> on each of the objects involved in the experiments and <bold>(B)</bold> on each session for all participants. The session number also reflects the corresponding number of objects present in the scene.</p>
</caption>
<graphic xlink:href="frobt-11-1346714-g003.tif"/>
</fig>
</sec>
<sec id="s5-5-2">
<title>5.5.2 Per session performance</title>
<p>
<xref ref-type="fig" rid="F3">Figure 3A</xref> depicts the accuracy levels of the overall pipeline in various sessions. The performance of the system slightly decreases for higher numbers of sessions. This is reasonable because, in those cases, the number of objects increases; thus, the table becomes more cluttered. However, the accuracy level is still acceptable (around 70%) even with the most cluttered scenes, showing that this is not a limitation of the proposed system.</p>
</sec>
<sec id="s5-5-3">
<title>5.5.3 Distractors</title>
<p>We investigate the impact of distracting objects on system performance, selecting a sample object (i.e., <italic>Pringles</italic>) as a distractor. Although participants were not instructed to focus on this object, our <italic>Object Detection</italic> module is trained to detect it. Our objective is to assess whether the presence of the distractor hinders the accurate identification of the target object. The results indicate that prediction errors occur in only about <inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>3% of frames with the distracting object, suggesting it is not a significant limitation.</p>
</sec>
<sec id="s5-5-4">
<title>5.5.4 Distance-based performance</title>
<p>A further analysis was conducted to evaluate the system&#x2019;s accuracy while systematically varying the distance between objects from 0 cm to 100 cm. Our method achieved <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mn>74</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> accuracy at 0 cm and over <inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mn>98</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> accuracy when objects were separated by more than 60 cm. The <xref ref-type="sec" rid="s13">Supplementary Material</xref> provides more details.</p>
</sec>
<sec id="s5-5-5">
<title>5.5.5 Real-time feed performance</title>
<p>The <xref ref-type="bibr" rid="B14">Chong et al. (2020)</xref> architecture, initially burdened by high latency due to reloading the model for each input frame, resulted in less than 5 fps output speed when integrated with our proposed system. To improve real-time performance, we separated model initialization from the code, initializing it only once. This adjustment boosted the output frame rate to 8 fps, deemed experimentally sufficient as the humanoid iCub&#x2019;s dynamics are slower than the camera frame rate.</p>
</sec>
<sec id="s5-5-6">
<title>5.5.6 Edge case performance</title>
<p>We assess the robustness of our system by conducting experiments on edge cases, including scenarios where the human partner is positioned at an angle relative to the robot and objects placed on the line of sight. These experiments yielded an overall accuracy level of <inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:mn>75</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. For more details, refer to the <xref ref-type="sec" rid="s13">Supplementary Material</xref>.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>We presented a learning system for detecting human attention toward objects in the scene. Our method combined an online object detection algorithm with a network for gaze estimation conditioned on the estimation of the human pose. We demonstrated its effectiveness through an extensive experimental analysis using the iCub robot. Our results indicated that integrating face detection, human attention prediction, and online object detection in our pipeline enables the robot to perceive and interpret human gaze within its environment. Such an achievement promises to enhance the robot&#x2019;s social awareness and responsiveness, allowing for more natural interactions in social robotics, which makes it well-suited to be used in applications such as assistant tutoring, robot-assisted therapies, and interaction with children with autism spectrum disorder (<xref ref-type="bibr" rid="B2">Alabdulkareem et al., 2022</xref>; <xref ref-type="bibr" rid="B38">Yousif, 2020</xref>; <xref ref-type="bibr" rid="B7">Calderita et al., 2014</xref>). The pipeline and dataset presented establish the foundation for our ongoing efforts to enhance iCub&#x2019;s collaborative task capabilities by integrating diverse social cues within a multimodal architecture. Our forthcoming endeavors will focus on integrating a segmentation layer to optimize system performance in more complex scenes (e.g., highly cluttered scenarios or the presence of non-convex objects). Another considered direction is to include out-of-frame target detection to identify when humans are not focused on the preferred task.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation. The code, learning models, and dataset can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/hsp-iit/online-attentive-object-detection">https://github.com/hsp-iit/online-attentive-object-detection</ext-link>.</p>
</sec>
<sec id="s8">
<title>Ethics statement</title>
<p>Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec id="s9">
<title>Author contributions</title>
<p>SH: data curation, investigation, methodology, software, validation, writing&#x2013;original draft, and writing&#x2013;review and editing. EM: conceptualization, data curation, investigation, methodology, software, supervision, validation, writing&#x2013;original draft, and writing&#x2013;review and editing. ML: conceptualization, data curation, investigation, methodology, software, supervision, validation, writing&#x2013;original draft, and writing&#x2013;review and editing. LN: conceptualization, funding acquisition, resources, supervision, and writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work received funding from the Italian National Institute for Insurance against Accidents at Work (INAIL) ergoCub Project and the project Fit for Medical Robotics (Fit4MedRob)&#x2013;PNRR MUR Cod. PNC0000007&#x2013;CUP: B53C22006960001.</p>
</sec>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frobt.2024.1346714/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frobt.2024.1346714/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Video1.mp4" id="SM2" mimetype="application/mp4" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn2">
<label>1</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://www.intelrealsense.com/depth-camera-d415/">https://www.intelrealsense.com/depth-camera-d415/</ext-link>
</p>
</fn>
<fn id="fn3">
<label>2</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://github.com/tzutalin/labelImg">https://github.com/tzutalin/labelImg</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Admoni</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Srinivasa</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Predicting user intent through eye gaze for shared autonomy</article-title>,&#x201d; in <source>2016 AAAI fall symposium series</source>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alabdulkareem</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alhakbani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Al-Nafjan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A systematic review of research on robot-assisted therapy for children with autism</article-title>. <source>Sensors</source> <volume>22</volume>, <fpage>944</fpage>. <pub-id pub-id-type="doi">10.3390/s22030944</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Allison</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Puce</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>McCarthy</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Social perception from visual cues: role of the sts region</article-title>. <source>Trends cognitive Sci.</source> <volume>4</volume>, <fpage>267</fpage>&#x2013;<lpage>278</lpage>. <pub-id pub-id-type="doi">10.1016/s1364-6613(00)01501-1</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Athavale</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Motati</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Kalahasty</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <source>One eye is all you need: lightweight ensembles for gaze estimation with single encoders</source>. <comment>arXiv preprint arXiv:2211.11936</comment>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Babel</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kraus</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kraus</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wagner</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Minker</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Small talk with a robot? the impact of dialog content, talk initiative, and gaze behavior of a social robot on trust, acceptance, and proximity</article-title>. <source>Int. J. Soc. Robotics</source> <volume>13</volume>, <fpage>1485</fpage>&#x2013;<lpage>1498</lpage>. <pub-id pub-id-type="doi">10.1007/s12369-020-00730-0</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bertasius</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>S. X.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <source>First person action-object detection with egonet</source>. <comment>arXiv preprint arXiv:1603.04908</comment>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calderita</surname>
<given-names>L. V.</given-names>
</name>
<name>
<surname>Manso</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Bustos</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Su&#xe1;rez-Mej&#xed;as</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fern&#xe1;ndez</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bandera</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Therapist: towards an autonomous socially interactive robot for motor and neurorehabilitation therapies for children</article-title>. <source>JMIR rehabilitation assistive Technol.</source> <volume>1</volume>, <fpage>e3151</fpage>. <pub-id pub-id-type="doi">10.2196/rehab.3151</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Calli</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Walsman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Srinivasa</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Dollar</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>The ycb object and model set: towards common benchmarks for manipulation research</article-title>,&#x201d; in <conf-name>2015 international conference on advanced robotics (ICAR)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>510</fpage>&#x2013;<lpage>517</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hidalgo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Simon</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>S.-E.</given-names>
</name>
<name>
<surname>Sheikh</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Openpose: realtime multi-person 2d pose estimation using part affinity fields</article-title>. <source>IEEE Trans. pattern analysis Mach. Intell.</source> <volume>43</volume>, <fpage>172</fpage>&#x2013;<lpage>186</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2019.2929257</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Simon</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>S.-E.</given-names>
</name>
<name>
<surname>Sheikh</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Realtime multi-person 2d pose estimation using part affinity fields</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>, <fpage>7291</fpage>&#x2013;<lpage>7299</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ceola</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Maiettini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pasquale</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rosasco</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Fast object segmentation learning with kernel-based methods for robotics</article-title>,&#x201d; in <conf-name>2021 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>13581</fpage>&#x2013;<lpage>13588</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA48506.2021.9561758</pub-id>
<volume>28</volume>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sato</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Gaze estimation by exploring two-eye asymmetry</article-title>. <source>IEEE Trans. Image Process.</source> <volume>29</volume>, <fpage>5259</fpage>&#x2013;<lpage>5272</lpage>. <pub-id pub-id-type="doi">10.1109/tip.2020.2982828</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chevalier</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kompatsiari</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ciardo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wykowska</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Examining joint attention with the use of humanoid robots-a new approach to study fundamental mechanisms of social cognition</article-title>. <source>Psychonomic Bull. and Rev.</source> <volume>27</volume>, <fpage>217</fpage>&#x2013;<lpage>236</lpage>. <pub-id pub-id-type="doi">10.3758/s13423-019-01689-4</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chong</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ruiz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Rehg</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Detecting attended visual targets in video</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>, <fpage>5396</fpage>&#x2013;<lpage>5406</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Doshi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Trivedi</surname>
<given-names>M. M.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Head and gaze dynamics in visual attention and context learning</article-title>,&#x201d; in <conf-name>2009 IEEE Computer Society Conference on Computer Vision and Pattern Recognition Workshops</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>77</fpage>&#x2013;<lpage>84</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fischer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Demiris</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Rt-gene: real-time eye gaze estimation in natural environments</article-title>,&#x201d; in <conf-name>Proceedings of the European conference on computer vision (ECCV)</conf-name>, <fpage>334</fpage>&#x2013;<lpage>352</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Furnari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Battiato</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Grauman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Farinella</surname>
<given-names>G. M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Next-active-object prediction from egocentric videos</article-title>. <source>J. Vis. Commun. Image Represent.</source> <volume>49</volume>, <fpage>401</fpage>&#x2013;<lpage>411</lpage>. <pub-id pub-id-type="doi">10.1016/j.jvcir.2017.10.004</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Holman</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Anwar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tec</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hart</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Watch where you&#x2019;re going! gaze and head orientation as predictors for social robot navigation</article-title>,&#x201d; in <conf-name>2021 IEEE International Conference on Robotics and Automation (ICRA)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>3553</fpage>&#x2013;<lpage>3559</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kurylo</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Wilson</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Using human eye gaze patterns as indicators of need for assistance from a socially assistive robot</article-title>,&#x201d; in <conf-name>Proceedings 11 Social Robotics: 11th International Conference, ICSR 2019</conf-name>, <conf-loc>Madrid, Spain</conf-loc>, <conf-date>November 26&#x2013;29, 2019</conf-date> (<publisher-name>Springer</publisher-name>), <fpage>200</fpage>&#x2013;<lpage>210</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Y. J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Visual instruction tuning</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>36</volume>. <pub-id pub-id-type="doi">10.48550/arXiv.2304.08485</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lombardi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maiettini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>De Tommaso</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wykowska</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Toward an attentive robotic architecture: learning-based mutual gaze estimation in human&#x2013;robot interaction</article-title>. <source>Front. Robotics AI</source> <volume>9</volume>, <fpage>770165</fpage>. <pub-id pub-id-type="doi">10.3389/frobt.2022.770165</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lombardi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maiettini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Tikhanoff</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022b</year>). &#x201c;<article-title>Icub knows where you look: exploiting social cues for interactive object detection learning</article-title>,&#x201d; in <conf-name>2022 IEEE-RAS 21st International Conference on Humanoid Robots (Humanoids)</conf-name>, <fpage>480</fpage>&#x2013;<lpage>487</lpage>. <pub-id pub-id-type="doi">10.1109/Humanoids53995.2022.10000163</pub-id>
<volume>30</volume>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maesako</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Koike</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Measurement of coordination of eye and head movements by sensor of terrestrial magnetism</article-title>. <source>Jpn. J. Physiological Psychol. Psychophysiol.</source> <volume>11</volume>, <fpage>69</fpage>&#x2013;<lpage>76</lpage>. <pub-id pub-id-type="doi">10.5674/jjppp1983.11.69</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Maiettini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pasquale</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rosasco</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Interactive data collection for deep learning object detectors on humanoid robots</article-title>,&#x201d; in <conf-name>2017 IEEE-RAS 17th International Conference on Humanoid Robotics (Humanoids)</conf-name>, <fpage>862</fpage>&#x2013;<lpage>868</lpage>. <pub-id pub-id-type="doi">10.1109/HUMANOIDS.2017.8246973</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maiettini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pasquale</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rosasco</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>On-line object detection: a robotics challenge</article-title>. <source>Aut. Robots</source> <volume>44</volume>, <fpage>739</fpage>&#x2013;<lpage>757</lpage>. <pub-id pub-id-type="doi">10.1007/s10514-019-09894-9</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Maiettini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pasquale</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Tikhanoff</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Rosasco</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019b</year>). &#x201c;<article-title>A weakly supervised strategy for learning object detection on a humanoid robot</article-title>,&#x201d; in <conf-name>2019 IEEE-RAS 19th International Conference on Humanoid Robots (Humanoids)</conf-name>, <fpage>194</fpage>&#x2013;<lpage>201</lpage>. <pub-id pub-id-type="doi">10.1109/Humanoids43949.2019.9035067</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Maiettini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Tikhanoff</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Weakly-supervised object detection learning through human-robot interaction</article-title>,&#x201d; in <conf-name>2020 IEEE-RAS 20th International Conference on Humanoid Robots (Humanoids)</conf-name>, <fpage>392</fpage>&#x2013;<lpage>399</lpage>. <pub-id pub-id-type="doi">10.1109/HUMANOIDS47582.2021.9555781</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Melvill Jones</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Guitton</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Berthoz</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>Changing patterns of eye-head coordination during 6 h of optically reversed vision</article-title>. <source>Exp. Brain Res.</source> <volume>69</volume>, <fpage>531</fpage>&#x2013;<lpage>544</lpage>. <pub-id pub-id-type="doi">10.1007/bf00247307</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Metta</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Fitzpatrick</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Yarp: yet another robot platform</article-title>. <source>Int. J. Adv. Robotic Syst.</source> <volume>3</volume>, <fpage>8</fpage>. <pub-id pub-id-type="doi">10.5772/5761</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Metta</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Natale</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Nori</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sandini</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Vernon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fadiga</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>The icub humanoid robot: an open-systems platform for research in cognitive development</article-title>. <source>Neural Netw.</source> <volume>23</volume>, <fpage>1125</fpage>&#x2013;<lpage>1134</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2010.08.010</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Palinko</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Rea</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sandini</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sciutti</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Eye gaze tracking for a humanoid robot</article-title>,&#x201d; in <conf-name>2015 IEEE-RAS 15th International Conference on Humanoid Robots (Humanoids)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>318</fpage>&#x2013;<lpage>324</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pelphrey</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Singerman</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Allison</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>McCarthy</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Brain activation evoked by perception of gaze shifts: the influence of context</article-title>. <source>Neuropsychologia</source> <volume>41</volume>, <fpage>156</fpage>&#x2013;<lpage>170</lpage>. <pub-id pub-id-type="doi">10.1016/s0028-3932(02)00146-x</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Recasens</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vondrick</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Khosla</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Torralba</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Following gaze in video</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE International Conference on Computer Vision</conf-name>, <fpage>1435</fpage>&#x2013;<lpage>1443</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Saran</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Majumdar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Short</surname>
<given-names>E. S.</given-names>
</name>
<name>
<surname>Thomaz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Niekum</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Human gaze following for human-robot interaction</article-title>,&#x201d; in <conf-name>2018 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>8615</fpage>&#x2013;<lpage>8621</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ververas</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gkagkos</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Christos Doukas</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zafeiriou</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <source>3dgazenet: generalizing gaze estimation with weak-supervision from synthetic views</source>. <comment>arXiv e-prints , arXiv&#x2013;2212</comment>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.-G.</given-names>
</name>
<name>
<surname>Sung</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Study on eye gaze estimation</article-title>. <source>IEEE Trans. Syst. Man, Cybern. Part B Cybern.</source> <volume>32</volume>, <fpage>332</fpage>&#x2013;<lpage>350</lpage>. <pub-id pub-id-type="doi">10.1109/tsmcb.2002.999809</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yousif</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Humanoid robot as assistant tutor for autistic children</article-title>. <source>Int. J. Comput. Appl. Sci.</source> <volume>8</volume> (<issue>2</issue>) .</citation>
</ref>
</ref-list>
</back>
</article>