<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">730317</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2021.730317</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>ExGenNet: Learning to Generate Robotic Facial Expression Using Facial Expression Recognition</article-title>
<alt-title alt-title-type="left-running-head">Rawal et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">ExGenNet: Robotic Facial Expression Generation</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Rawal</surname>
<given-names>Niyati</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1235842/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Koert</surname>
<given-names>Dorothea</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/808235/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Turan</surname>
<given-names>Cigdem</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/301714/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kersting</surname>
<given-names>Kristian</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/321053/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Peters</surname>
<given-names>Jan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/760887/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Stock-Homburg</surname>
<given-names>Ruth</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Chair for Marketing and Human Resource Management, Department of Law and Economics, Technical University of Darmstadt</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Intelligent Autonomous Systems, Department of Computer Science, Technical University of Darmstadt</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>AI and Machine Learning Group, Department of Computer Science, Technical University of Darmstadt</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Centre for Cognitive Science and Hessian Center for AI (hessian.AI)</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Leap in Time&#x2014;Work Life Research Institute</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/138031/overview">Basilio Sierra</ext-link>, University of the Basque Country, Spain</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/582683/overview">Mirko Rakovic</ext-link>, University of Novi Sad, Serbia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/525016/overview">Daniele Cafolla</ext-link>, Istituto Neurologico Mediterraneo Neuromed (IRCCS), Italy</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Niyati Rawal, <email>rawal.niyati@gmail.com</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Humanoid Robotics, a section of the journal Frontiers in Robotics and&#x20;AI</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>01</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>8</volume>
<elocation-id>730317</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>06</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>11</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Rawal, Koert, Turan, Kersting, Peters and Stock-Homburg.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Rawal, Koert, Turan, Kersting, Peters and Stock-Homburg</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>The ability of a robot to generate appropriate facial expressions is a key aspect of perceived sociability in human-robot interaction. Yet many existing approaches rely on the use of a set of fixed, preprogrammed joint configurations for expression generation. Automating this process provides potential advantages to scale better to different robot types and various expressions. To this end, we introduce ExGenNet, a novel deep generative approach for facial expressions on humanoid robots. ExGenNets connect a generator network to reconstruct simplified facial images from robot joint configurations with a classifier network for state-of-the-art facial expression recognition. The robots&#x2019; joint configurations are optimized for various expressions by backpropagating the loss between the predicted expression and intended expression through the classification network and the generator network. To improve the transfer between human training images and images of different robots, we propose to use extracted features in the classifier as well as in the generator network. Unlike most studies on facial expression generation, ExGenNets can produce multiple configurations for each facial expression and be transferred between robots. Experimental evaluations on two robots with highly human-like faces, Alfie (Furhat Robot) and the android robot Elenoide, show that ExGenNet can successfully generate sets of joint configurations for predefined facial expressions on both robots. This ability of ExGenNet to generate realistic facial expressions was further validated in a pilot study where the majority of human subjects could accurately recognize most of the generated facial expressions on both the robots.</p>
</abstract>
<kwd-group>
<kwd>facial expression generation</kwd>
<kwd>humanoid robots</kwd>
<kwd>facial expression recognition</kwd>
<kwd>neural networks</kwd>
<kwd>gradient descent</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Perceived sociability is an important aspect in human-robot interaction (HRI), and users want robots to behave in a friendly and emotionally intelligent manner (<xref ref-type="bibr" rid="B21">Nicolescu and Mataric, 2001</xref>; <xref ref-type="bibr" rid="B12">Hoffman and Breazeal, 2006</xref>; <xref ref-type="bibr" rid="B26">Ray et&#x20;al., 2008</xref>; <xref ref-type="bibr" rid="B10">de Graaf et&#x20;al., 2016</xref>). Studies indicate that in any interaction, 7% of the affective information is conveyed through words, 38% is conveyed through tone, and 55% is conveyed through facial expressions (<xref ref-type="bibr" rid="B19">Mehrabian, 1968</xref>). This makes facial expressions an indispensable mode of communicating affective information and, subsequently, generating appropriate and realistic facial expressions, which can be perceived by humans, and a key ability for humanoid robots.</p>
<p>While methods for automated facial expression recognition have been a research field in human-robot interaction for several years (see <xref ref-type="bibr" rid="B16">Li and Deng (2020</xref>); <xref ref-type="bibr" rid="B7">Canedo and Neves (2019</xref>), for overviews), facial expression generation for humanoid robots is a younger line of research (<xref ref-type="bibr" rid="B6">Breazeal, 2003</xref>; <xref ref-type="bibr" rid="B14">Kim et&#x20;al., 2006</xref>; <xref ref-type="bibr" rid="B11">Ge et&#x20;al., 2008</xref>; <xref ref-type="bibr" rid="B13">Horii et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B18">Meghdari et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B27">Silva et&#x20;al., 2016</xref>). Moreover, most of the existing studies use preprogrammed joint configurations, for example, adjusting the servo motors movement for the eyelids and the mouth to form the basic expressions in a hand-coded manner (<xref ref-type="bibr" rid="B6">Breazeal, 2003</xref>; <xref ref-type="bibr" rid="B14">Kim et&#x20;al., 2006</xref>; <xref ref-type="bibr" rid="B11">Ge et&#x20;al., 2008</xref>; <xref ref-type="bibr" rid="B3">Bennett and Sabanovic, 2014</xref>; <xref ref-type="bibr" rid="B27">Silva et&#x20;al., 2016</xref>). While this allows studying human reactions to robot expressions, it requires hand-tuning for every individual robot and re-programming in case of hardware adaptations on a robot&#x2019;s face. In contrast, the automated generation of robot facial expressions could alleviate the &#x201c;hard-coding&#x201d; of expressions, thereby making it easier to scale to different robots in a more principled manner. Still, there are only a few studies that learn different configurations for expressions automatically (<xref ref-type="bibr" rid="B5">Breazeal et&#x20;al., 2005</xref>; <xref ref-type="bibr" rid="B13">Horii et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B8">Churamani et&#x20;al., 2018</xref>), and most of the studies only learn to generate a single configuration for each facial expression. Humans, however, usually exhibit a variety of different expressive ways instead of a single configuration per facial expression. Such high expressiveness can be tedious to hand-tune, which can potentially be achieved easier in a generative setting.</p>
<p>In this article, we propose a novel approach to automatically learn multiple joint configurations for facial expressions on humanoid robots, called ExGenNet. In particular, we suggest utilizing state-of-the-art deep network approaches for image-based human facial expression recognition as feedback to train a pipeline of robot facial expression generation (See <xref ref-type="fig" rid="F1">Figure&#x20;1</xref> for an overview). We evaluate ExGenNet on the two robots Elenoide and Alfie (Furhat Robot) with highly human-like faces, where we make use of facial feature extractors for smooth transfer between the different robot types. The learned facial expressions are additionally evaluated in a pilot study where we investigate how images of the robots displaying the autogenerated expressions are perceived by humans.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>We propose a novel approach to automatically learn multiple joint configurations for facial expressions on highly humanoid robots, such as Alfie <bold>(left)</bold> and Elenoide <bold>(right)</bold>. For generation of multiple facial expressions we utilize a classifier based on simplified images of extracted facial features from a mixed dataset of human and robot training pictures.</p>
</caption>
<graphic xlink:href="frobt-08-730317-g001.tif"/>
</fig>
<p>To summarize, our main contributions are threefold. First, ExGenNet introduces a novel automated way of learning multiple joint configurations per expression via a gradient-based minimization of the loss between the intended expression and the expression predicted by the trained classifier. Second, ExGenNet uses facial features as a shared representation between human training data and different robots. Finally, we present insights on expression generation on highly humanoid robots and the way they are perceived by humans.</p>
<p>The rest of the article is structured as follows. In <xref ref-type="sec" rid="s2">Section 2</xref>, we give a short overview of related work followed by the details of our algorithm in <xref ref-type="sec" rid="s3">Section 3</xref>. In <xref ref-type="sec" rid="s4">Section 4</xref>, we present the results of our experimental evaluations on two robots with highly human-like faces. Finally, we conclude and discuss future work in <xref ref-type="sec" rid="s5">Section&#x20;5</xref>.</p>
</sec>
<sec id="s2">
<title>2 Related Work</title>
<p>Facial expression recognition has been well studied in human-robot interaction (HRI) (e.g., <xref ref-type="bibr" rid="B9">Cid et&#x20;al., 2014</xref>; <xref ref-type="bibr" rid="B18">Meghdari et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B28">Simul et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B4">Bera et&#x20;al., 2019</xref>). As deep learning methods have become popular, facial expression recognition nowadays mostly consists of preprocessing the facial images and directly feeding them into deep networks to predict an output (<xref ref-type="bibr" rid="B16">Li and Deng, 2020</xref>). Among all approaches, Convolutional Neural Networks (CNNs) are widely used for performing facial expression recognition during HRI (see <xref ref-type="bibr" rid="B25">Rawal and Stock-Homburg (2021</xref>) for an overview). In particular, CNNs are used for end-to-end recognition, i.e.,&#x20;given the input image, the output is directly predicted by the network. ExGenNets employ CNNs for the expression classification part of our pipeline.</p>
<p>In the field of computer vision, there are many works that tackle facial expression generation within pictures of persons, other characters, or even animals. For instance, <xref ref-type="bibr" rid="B22">Noh and Neumann (2006)</xref> transferred motion vectors from a source to a target face model. Here, the target face model could have completely different geometric proportions and mesh structures, compared to the source face model. <xref ref-type="bibr" rid="B23">Pighin et&#x20;al. (2006)</xref> created photorealistic 3D facial models from photographs of a human, and further created continuous and realistic transitions between different facial expressions by morphing between different models. <xref ref-type="bibr" rid="B24">Pumarola et&#x20;al. (2019)</xref> implemented a Generative Adversarial Network (GAN) to animate a given image and render novel expressions in a continuum. While these approaches are capable of generating &#x201c;images&#x201d; of robots with different facial expressions, they are incapable of directly being transferred to robots since the joint angles need to be optimized to realize an expression.</p>
<p>This may explain why most approaches to facial expression generation on humanoid robots rely on hand-tuned expression generation. Only a few studies so far introduced automated ways for expression generation based on simple Neural Networks (<xref ref-type="bibr" rid="B5">Breazeal et&#x20;al., 2005</xref>), Restricted Boltzmann Machines (RBM) (<xref ref-type="bibr" rid="B13">Horii et&#x20;al., 2016</xref>), and Reinforcement Learning (RL) (<xref ref-type="bibr" rid="B8">Churamani et&#x20;al., 2018</xref>).</p>
<p>Specifically, <xref ref-type="bibr" rid="B5">Breazeal et&#x20;al. (2005)</xref> used neural networks to learn a direct mapping of human facial expressions onto a robot&#x2019;s joint space. <xref ref-type="bibr" rid="B13">Horii et&#x20;al. (2016)</xref> used an RBM to generate expressions on an iCub robot. The forward sampling in the RBM is to recognize the human counterpart&#x2019;s expression based on facial, audio, and gestural data during HRI, while the backward sampling is to generate facial, audio, and gestural data for the robot. <xref ref-type="bibr" rid="B8">Churamani et&#x20;al. (2018)</xref> generated expressions on their robot Nico using RL. They used an actor-critic network where the actor network receives the current mood affect vector of the robot and generates an action (LED configurations for the eyebrows and mouth) corresponding to this state. The action and the state are then fed into the critic network that predicts a Q-value for the state-action pair. The robot receives an award based on symmetry in wavelengths generated by the network.</p>
<p>All of these studies only generate single configurations per expression. In contrast, ExGenNets are able to generate a range of joint configurations per expression. Additionally, even though the robots used in related approaches are humanoid robots, their faces are not as human-like as the faces of the robots used in our experimental evaluation. Since these robots may start entering the uncanny valley (<xref ref-type="bibr" rid="B20">Mori et&#x20;al., 2012</xref>), we consider evaluation of the human perception of expressions generated on robots with highly human-like faces is particularly important.</p>
</sec>
<sec id="s3">
<title>3 ExGenNet: Expression Generation Network</title>
<p>Expression Generation Networks (ExGenNets) learn multiple joint configurations for previously defined facial expressions on humanoid robots. To achieve this, they combine a generator network, which reconstructs a simplified image of facial landmarks for a given configuration, together with a CNN-based expression classifier. <xref ref-type="fig" rid="F2">Figure&#x20;2</xref> summarizes the overall approach.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Network structure of the proposed ExGenNet (Expression Generation Network) and optimization procedure of the joint configuration <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. First, the generator network generated an image of facial features for a given joint configuration <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Next, we feed this image into an expression classifier that classifies the image into five categories: angry, happy, neutral, sad, and surprise. The cross-entropy loss between the predicted expression and desired expression is calculated. We find the value of the joint configuration (global minima) for which the cross-entropy loss is the minimum by performing a grid search. To find the local minima (mean and standard deviation), we backpropagate the cross-entropy loss through the classifier and the generator networks to update the joint configuration. The goal is to find the appropriate range of joint configuration for a given facial expression.</p>
</caption>
<graphic xlink:href="frobt-08-730317-g002.tif"/>
</fig>
<p>Let us now introduce ExGenNets in detail. In <xref ref-type="sec" rid="s3-1">Section 3.1</xref>, we present the details of the automated expression generation of facial landmarks. <xref ref-type="sec" rid="s3-2">Section 3.2</xref> introduces the facial feature-based expression classifier. Lastly, <xref ref-type="sec" rid="s3-3">Section 3.3</xref> explains the optimization of joint values for different expressions.</p>
<sec id="s3-1">
<title>3.1 Simplified Expression Image Generator</title>
<p>To learn a mapping between the robot joint configurations and facial expressions, we train a generator network to convert from joint angles to a simplified facial image<disp-formula id="e1">
<mml:math id="m4">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>image</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>robot</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <bold>
<italic>X</italic>
</bold>
<sub>image</sub> denotes the image of the facial features generated by the network, <italic>h</italic>
<sub>robot</sub> the generator network, and <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> the joint configuration of the robot. To reduce computational effort, we obtain simplified images by applying the dlib facial feature detector (<xref ref-type="bibr" rid="B15">King, 2009</xref>) on the full images of the robot and constructing a smaller and less detailed black and white image from this. An example of such a simplified image can be seen in <xref ref-type="fig" rid="F3">Figure&#x20;3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>An example of a simplified image showing the facial landmarks in black. The facial landmarks are detected using dlib (<xref ref-type="bibr" rid="B15">King, 2009</xref>). The simplified expression image generator generates images of this&#x20;kind.</p>
</caption>
<graphic xlink:href="frobt-08-730317-g003.tif"/>
</fig>
<p>For the generator network we use a CNN with five layers, which generates images of size 48 &#xd7; 48&#x20;&#xd7; 1. The network structure is given in <xref ref-type="table" rid="T1">Table&#x20;1</xref>. We created a dataset of the images of facial feature extraction at various values of joint configurations from the robot. Next, we trained the generator network <italic>h</italic>
<sub>robot</sub> to generate images of facial feature extraction from robot joint configuration. The generator network that we use is similar to the generator network in standard Generative Adversarial Networks (GANs), except we do not use a discriminator to train the images being generated. As the output consists of only simplified images of facial features, we found it is sufficient to train the generator network by reducing the mean squared error between the pixels of the actual output and the predicted output.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>CNN Network structure for generator.</p>
</caption>
<table>
<thead>
<tr>
<td align="left">Layer (type)</td>
<td align="center">Output shape</td>
<td align="center">Param &#x23;</td>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">dense (Dense)</td>
<td align="center">(None, 9,216)</td>
<td align="center">55,296/27,648</td>
</tr>
<tr>
<td align="left">batch_normalization (BatchNormalization)</td>
<td align="center">(None, 9,216)</td>
<td align="center">36,864</td>
</tr>
<tr>
<td align="left">leaky_relu (LeakyReLU)</td>
<td align="center">(None, 9,216)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">reshape (Reshape)</td>
<td align="center">(None, 6, 6, 256)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">conv2d_transpose (Conv2DTranspose)</td>
<td align="center">(None, 6, 6, 128)</td>
<td align="center">294,912</td>
</tr>
<tr>
<td align="left">batch_normalization_1 (BatchNormalization)</td>
<td align="center">(None, 6, 6, 128)</td>
<td align="center">512</td>
</tr>
<tr>
<td align="left">leaky_relu_1 (LeakyReLU)</td>
<td align="center">(None, 6, 6, 128)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">conv2d_transpose_1 (Conv2DTranspose)</td>
<td align="center">(None, 12, 12, 64)</td>
<td align="center">73,728</td>
</tr>
<tr>
<td align="left">batch_normalization_2 (BatchNormalization)</td>
<td align="center">(None, 12, 12, 64)</td>
<td align="center">256</td>
</tr>
<tr>
<td align="left">leaky_relu_2 (LeakyReLU)</td>
<td align="center">(None, 12, 12, 64)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">conv2d_transpose_2 (Conv2DTranspose)</td>
<td align="center">(None, 24, 24, 32)</td>
<td align="center">18,432</td>
</tr>
<tr>
<td align="left">batch_normalization_3 (BatchNormalization)</td>
<td align="center">(None, 24, 24, 32)</td>
<td align="center">128</td>
</tr>
<tr>
<td align="left">leaky_relu_3 (LeakyReLU)</td>
<td align="center">(None, 24, 24, 32)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">conv2d_transpose_3 (Conv2DTranspose)</td>
<td align="center">(None, 48, 48, 1)</td>
<td align="center">288</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>3.2&#x20;Feature-Based Expression Classifier</title>
<p>Most studies that perform facial expression recognition directly train on facial images (<xref ref-type="bibr" rid="B2">Barros et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B1">Ahmed et&#x20;al., 2019</xref>). In contrast, we train the Convolutional Neural Network (CNN) on simplified versions of the training images generated from landmarks detected via dlib (<xref ref-type="bibr" rid="B15">King, 2009</xref>). As training images, we used the KDEF dataset (<xref ref-type="bibr" rid="B17">Lundqvist et&#x20;al., 1998</xref>) and additional hand-labeled training images for our robots Alfie and Elenoide. We used data augmentation to slightly translate the training images both vertically and horizontally so that the faces are visible completely. We also evaluate the models trained with different combinations of data, i.e.,&#x20;only human facial expressions, only robot facial expressions, and both combined. The use of the extracted features, i.e. landmark points, makes the classifier hereby less sensitive to environmental changes such as lighting conditions and generalizes better between different robot types and human images. From the set of labeled training images, the classifier should learn the mapping from the simplified facial images to discrete predefined expressions<disp-formula id="e2">
<mml:math id="m6">
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>expression</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>classifier</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>image</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>with the expression classifier <italic>f</italic>
<sub>classifier</sub> and the simplified image of the extracted facial features <bold>X</bold>
<sub>image</sub> of size 48 &#xd7; 48&#x20;&#xd7; 1. For the expressions classifier, we used CNN with four convolutional layers and one fully-connected layer. The network structure is given in <xref ref-type="table" rid="T2">Table&#x20;2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>CNN Network structure for expression classifier.</p>
</caption>
<table>
<thead>
<tr>
<td align="left">Layer (type)</td>
<td align="center">Output shape</td>
<td align="center">Param &#x23;</td>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">conv2d (Conv2D)</td>
<td align="center">(None, 46, 46, 32)</td>
<td align="center">320</td>
</tr>
<tr>
<td align="left">conv2d_1 (Conv2D)</td>
<td align="center">(None, 44, 44, 64)</td>
<td align="center">18,496</td>
</tr>
<tr>
<td align="left">max_pooling2d (MaxPooling2D)</td>
<td align="center">(None, 22, 22, 64)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">dropout (Dropout)</td>
<td align="center">(None, 22, 22, 64)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">conv2d_2 (Conv2D)</td>
<td align="center">(None, 20, 20, 128)</td>
<td align="center">73,856</td>
</tr>
<tr>
<td align="left">max_pooling2d_1 (MaxPooling2D)</td>
<td align="center">(None, 10, 10, 128)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">conv2d_3 (Conv2D)</td>
<td align="center">(None, 8, 8, 128)</td>
<td align="center">147,584</td>
</tr>
<tr>
<td align="left">max_pooling2d_2 (MaxPooling2D)</td>
<td align="center">(None, 4, 4, 128)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">dropout_1 (Dropout)</td>
<td align="center">(None, 4, 4, 128)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">flatten (Flatten)</td>
<td align="center">(None, 2048)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">dense_1 (Dense)</td>
<td align="center">(None, 1,024)</td>
<td align="center">2,098,176</td>
</tr>
<tr>
<td align="left">dropout_2 (Dropout)</td>
<td align="center">(None, 1,024)</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">dense_2 (Dense)</td>
<td align="center">(None, 5)</td>
<td align="center">5,125</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<title>3.3 Automated Expression Generation</title>
<p>Here, we explain in detail how we automatically calculate and obtain the values of joint configurations <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> by performing grid search and gradient descent. To automatically find a set of joint configurations that generate a desired facial expression <italic>y</italic>
<sub>
<italic>j</italic>
</sub> on the robot, we minimize the cross-entropy loss <italic>L</italic> between the predicted expression of the classifier <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>classifier</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>robot</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> and the desired expression<disp-formula id="e3">
<mml:math id="m9">
<mml:mi>L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>with <italic>N</italic> predefined facial expressions. To find stable joint configurations, we determined and averaged the loss for five generator networks and five classifier networks.</p>
<p>Specifically, we first find the global minima for <inline-formula id="inf7">
<mml:math id="m10">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> values in the whole range. To this end, we consider <inline-formula id="inf8">
<mml:math id="m11">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> values at discrete steps and find the combination of joint angles for which the loss is minimal for each expression. Then, we find the global minima. We assume a range for values of <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> instead of just a single joint configuration per expression by considering <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> where <inline-formula id="inf11">
<mml:math id="m14">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the mean and <inline-formula id="inf12">
<mml:math id="m15">
<mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the standard deviation. We used&#x20;reparameterization to sample from <inline-formula id="inf13">
<mml:math id="m16">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf14">
<mml:math id="m17">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> by considering <inline-formula id="inf15">
<mml:math id="m18">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x0298;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and correspondingly propagate the gradients to minimize the loss via gradient descent<disp-formula id="e4">
<mml:math id="m19">
<mml:msub>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>L</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>classifier</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>robot</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(4)</label>
</disp-formula>using element-wise multiplication &#x0298; and Gaussian noise <inline-formula id="inf16">
<mml:math id="m20">
<mml:mrow>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>The mean and the variance are updated by gradient descent<disp-formula id="e5">
<mml:math id="m21">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(5)</label>
</disp-formula>using a learning rate <italic>&#x3b1;</italic> and computing the gradients by <inline-formula id="inf17">
<mml:math id="m22">
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf18">
<mml:math id="m23">
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x0298;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<xref ref-type="other" rid="alg1">Algorithm 1</xref> summarizes the overall approach. Here, <italic>f</italic> is the expression-classifier and <italic>h</italic> is the image generated for Alfie&#x2019;s and Elenoide&#x2019;s face. The loss L and gradients <inline-formula id="inf19">
<mml:math id="m24">
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo>&#x20d7;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf20">
<mml:math id="m25">
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo>&#x20d7;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x0298;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mo>&#x20d7;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are averaged over minibatch.</p>
<p>
<statement content-type="algorithm" id="alg1">
<label>Algorithm 1</label>
<p>Generating multiple configurations for various expressions.</p>
</statement>
</p>
<p>
<inline-graphic xlink:href="frobt-08-730317-fx1.tif"/>
</p>
</sec>
</sec>
<sec id="s4">
<title>4 Experimental Evaluation</title>
<p>We evaluated our proposed method on two highly humanoid robots, named Alfie (<xref ref-type="fig" rid="F4">Figure&#x20;4</xref>) and Elenoide (<xref ref-type="fig" rid="F5">Figure&#x20;5</xref>). First, we present the results of expression generation using our novel ExGenNet approach in <xref ref-type="sec" rid="s4-1">Section 4.1</xref>. Additionally, we report in <xref ref-type="sec" rid="s4-2">Section 4.2</xref> the results of a pilot study, where we evaluated how the generated expressions are perceived by humans. Finally, we discuss the remaining limitations of our method in <xref ref-type="sec" rid="s4-3">Section&#x20;4.3</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Alfie (real Furhat Robot) displaying <bold>(A)</bold> Angry, <bold>(B)</bold> Happy, <bold>(C)</bold> Neutral, <bold>(D)</bold> Sad, and <bold>(E)</bold> Surprise expressions at <inline-formula id="inf34">
<mml:math id="m39">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> respectively. Here, <inline-formula id="inf35">
<mml:math id="m40">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the mean and <inline-formula id="inf36">
<mml:math id="m41">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the standard deviation. We obtain the <inline-formula id="inf37">
<mml:math id="m42">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and the <inline-formula id="inf38">
<mml:math id="m43">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> values for various expressions from the ExGenNet by reparameterizing <inline-formula id="inf39">
<mml:math id="m44">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as <inline-formula id="inf40">
<mml:math id="m45">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="bold-italic">&#x0298;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</caption>
<graphic xlink:href="frobt-08-730317-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Elenoide displaying <bold>(A)</bold> Happy, <bold>(B)</bold> Neutral, and <bold>(C)</bold> Surprise expressions at <inline-formula id="inf41">
<mml:math id="m46">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> respectively. Here, <inline-formula id="inf42">
<mml:math id="m47">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the mean and <inline-formula id="inf43">
<mml:math id="m48">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the standard deviation. We obtain the <inline-formula id="inf44">
<mml:math id="m49">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and the <inline-formula id="inf45">
<mml:math id="m50">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> values for various expressions from the ExGenNet by reparameterizing <inline-formula id="inf46">
<mml:math id="m51">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as <inline-formula id="inf47">
<mml:math id="m52">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x0298;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</caption>
<graphic xlink:href="frobt-08-730317-g005.tif"/>
</fig>
<sec id="s4-1">
<title>4.1 Expression Generation on the Robots</title>
<p>We conducted experiments with the approach described in <xref ref-type="sec" rid="s3">Section 3</xref> on our robots Alfie and Elenoide. The code is written in Python using TensorFlow. Adam was used as an optimizer. For the grid search, we considered discrete intervals of 0.1 for each of the three joint angles for Elenoide and 0.2 for each of the six joint angles for Alfie. For fine-tuning the <inline-formula id="inf48">
<mml:math id="m53">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo>&#x20d7;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> values and obtaining a range for each expression, we initialized the mean to 0 and standard deviation to 0.25 for all joint angles <inline-formula id="inf49">
<mml:math id="m54">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo>&#x20d7;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> for both Elenoide and Alfie. We considered the following joints for Alfie: BLINK_LEFT, BLINK_RIGHT, BROW_UP_LEFT, BROW_UP_RIGHT, BROW_DOWN_LEFT, BROW_DOWN_RIGHT, SMILE_OPEN, PHONE_CH_J_SH, and PHONE_BIGAAH. For Elenoide, we considered the joints in the eyes, eyebrows, and mouth (mouth opening and mouth corner pull). Alfie is able to generate angry, happy, neutral, sad, and surprise expressions (see <xref ref-type="fig" rid="F4">Figure&#x20;4</xref>). Elenoide is able to generate only happy, neutral, and surprise expressions (see <xref ref-type="fig" rid="F5">Figure&#x20;5</xref>). Owing to restrictions in the degrees of freedom in the mouth and eyebrows, Elenoide is not able to generate negative facial expressions like sad and&#x20;angry.</p>
<p>To have consistent values for the joint configurations, we averaged the loss in <xref ref-type="disp-formula" rid="e4">Equation 4</xref> for five generator and five classifier models with different random initializations. The graphs showing the loss and accuracy for the five classifier models can be seen in <xref ref-type="fig" rid="F6">Figure&#x20;6</xref>. The epoch number where the loss is minimum is selected for the five classifier models that are then used in the ExGenNet for obtaining the values of the joint angles for various expressions.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Graphs showing the average loss and the average accuracy for the five classifier models trained with random seeds. The epoch number where the loss was minimum was selected for obtaining the values of the joint angles for each of the five classifier networks.</p>
</caption>
<graphic xlink:href="frobt-08-730317-g006.tif"/>
</fig>
<p>After generating the expressions for the two robots, we verified the facial expressions using the previously trained classifiers. For both Elenoide and Alfie, we tested the five classifiers to recognize the expressions for images obtained at <inline-formula id="inf50">
<mml:math id="m55">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. We also sample 10 images from <inline-formula id="inf51">
<mml:math id="m56">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf52">
<mml:math id="m57">
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and average the classification results. On Elenoide, the average accuracy for recognizing the expressions correctly was 81% for the five classifiers trained on human and robot dataset (see <xref ref-type="fig" rid="F7">Figure&#x20;7</xref>). For classifiers trained on only robot and only human datasets, the average accuracies were 69 and 52% respectively. In the case of Alfie, we trained the ExGenNet and obtained the mean and standard deviation for various expressions on the simulator. <xref ref-type="fig" rid="F4">Figure&#x20;4</xref> shows the final results on the real robots. For the classification results on the images of the real robot, the average accuracy was 70% for the five classifiers trained on the human and robot dataset. The classifiers trained on only human and only robot datasets had average accuracies of 43 and 30%, respectively. We also tested the classification results on Alfie&#x2019;s simulator. The average accuracy for the five classifiers trained on human and robot data combined was 75%, followed by the classifiers trained on only robot data 51%. The accuracy for the classifiers trained on only human data was 39%. For Elenoide, Alfie, and Alfie&#x2019;s simulator, the classifier that is trained on the combined dataset of humans and robots gives the best results. While the classifier trained on only robot data has a higher expression recognition rate for Elenoide and Alfie&#x2019;s simulator, the classifier trained on only human data has better results in the case of Alfie. This is because the classifier trained on only robot data consists of simplified face images of only Elenoide and Alfie&#x2019;s simulator. As Alfie and Alfie&#x2019;s simulator has a different rendering and the classifier trained on only robot data overfits to the data of Elenoide and Alfie&#x2019;s simulator, the classifier trained on only robot data performs worse than the classifier trained on human data for Alfie.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Comparison of classifier accuracy for Elenoide, Alfie (real robot), and Alfie&#x2019;s simulator when trained with combined human and robot dataset (blue), only robot dataset (orange), and only human dataset (green). Classifiers trained with combined human and robot dataset performed the best for all Elenoide, Alfie, and Alfie&#x2019;s simulator.</p>
</caption>
<graphic xlink:href="frobt-08-730317-g007.tif"/>
</fig>
</sec>
<sec id="s4-2">
<title>4.2 Human Perception of Generated Expressions</title>
<p>To validate the results further, we also conduct a pilot study to investigate how images of autogenerated expressions are perceived by humans. Here, we conducted three surveys, one for humans to be able to recognize the expressions generated on Elenoide (happy, neutral, and surprise) and another two for humans to be able to recognize the positive expressions (happy, neutral, and surprise) and the negative expressions (angry, neutral, and sad) on Alfie. In each survey, we randomly showed three images per expression and asked two questions. The first question is how positive or negative the expressions look and the second question is to recognize the expression in the image out of five categories: angry, happy, neutral, sad, and surprise. There is also an additional optional category for &#x201c;other&#x201d; expressions in case it seems that the expression in the image does not belong to any of the five categories. The first question is rated between 1 and 7, where 1 indicates that the expression looks negative and 7 indicates that the expression looks positive. The second question is rated on a five-point Likert scale, where 1 indicates strongly disagree and 5 indicates strongly agree. In between the survey, there was a screening question to check the participants&#x2019; attention. The screening question asked the participants to choose &#x201c;strongly agree&#x201d; on the same five-point Likert&#x20;scale.</p>
<p>We obtained responses from 30 participants for each survey. We did not consider the responses of the participants who answered the screening question incorrectly and ended up evaluating the results for 27 participants in the case of Elenoide, 29 participants to recognize the positive expressions for Alfie, and 27 participants to recognize the negative expressions for Alfie. Out of all the participants, 65% were male and 35% were female with an average age of 35&#xa0;years. We performed the Pearson&#x2019;s correlation test using the Scipy library in Python. First, the mean values of all the expressions for all participants on a five-point Likert scale were calculated. Next, the intended labels were assigned. For example, if the intended class is happy, the intended labels would be [1,5,1,1,1,1]. The Pearson&#x2019;s correlation test was performed to compare the classes chosen by humans with the intended classes. The Scipy library in Python returns both the Pearson&#x2019;s correlation coefficient and a two-tailed <italic>p</italic>-value for the Pearson&#x2019;s function. The Pearson&#x2019;s correlation coefficient and the two-tailed <italic>p</italic>-value are given in <xref ref-type="table" rid="T3">Table&#x20;3</xref> for various expressions.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Results of Pearson&#x2019;s correlation test for various expressions on Elenoide and Alfie.</p>
</caption>
<table>
<thead>
<tr>
<td align="left">Robot</td>
<td align="center">Expression</td>
<td align="center">Pearson&#x2019;s correlation coefficient</td>
<td align="center">
<italic>p</italic>-value</td>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Alfie</td>
<td align="left">Happy</td>
<td align="center">0.87</td>
<td align="center">p <inline-formula id="inf53">
<mml:math id="m58">
<mml:mo>&#x3c;</mml:mo>
</mml:math>
</inline-formula> 0.001</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Neutral</td>
<td align="center">0.68</td>
<td align="center">p <inline-formula id="inf54">
<mml:math id="m59">
<mml:mo>&#x3c;</mml:mo>
</mml:math>
</inline-formula> 0.001</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Surprise</td>
<td align="center">0.82</td>
<td align="center">p <inline-formula id="inf55">
<mml:math id="m60">
<mml:mo>&#x3c;</mml:mo>
</mml:math>
</inline-formula> 0.001</td>
</tr>
<tr>
<td align="left">Alfie</td>
<td align="left">Angry</td>
<td align="center">0.37</td>
<td align="center">0.04</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Neutral</td>
<td align="center">0.45</td>
<td align="center">0.01</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Sad</td>
<td align="center">&#x2212;0.17</td>
<td align="center">0.37</td>
</tr>
<tr>
<td align="left">Elenoide</td>
<td align="left">Happy</td>
<td align="center">0.63</td>
<td align="center">p <inline-formula id="inf56">
<mml:math id="m61">
<mml:mo>&#x3c;</mml:mo>
</mml:math>
</inline-formula> 0.001</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Neutral</td>
<td align="center">0.24</td>
<td align="center">0.19</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Surprise</td>
<td align="center">0.89</td>
<td align="center">p <inline-formula id="inf57">
<mml:math id="m62">
<mml:mo>&#x3c;</mml:mo>
</mml:math>
</inline-formula> 0.001</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We found that in the case of Alfie, all the positive expressions are positively correlated, happy being the most positively correlated, followed by surprise and thereafter neutral. In the case of negative expressions, sad was found to be negatively correlated. Alfie does not have a joint configuration in its mouth that can make the corners of its mouth turn downward in case of a sad expression, always forming a slight smile. Therefore, Alfie cannot express sadness the way a human does. Neutral and angry were found to be positively correlated. The <italic>p</italic>-value for all expressions except sad was less than 0.05. Therefore, the correlation coefficient for all expressions except sad is significant. In the case of Elenoide, all three expressions were positively correlated, &#x201c;surprise&#x201d; is the most positively correlated, followed by happy and neutral. For neutral, the <italic>p</italic>-value was greater than 0.05, implying that the result is not significant. While observing the data obtained from the participants, it was found that the neutral images of Elenoide were reported as both happy and neutral. This is probably because the corners of the mouth for Elenoide are always pulled upward even for neutral expressions, forming a&#x20;smile.</p>
</sec>
<sec id="s4-3">
<title>4.3 Discussion of Limitations</title>
<p>While our proposed approach successfully generated multiple joint configurations for facial expressions of the two robots, which were correctly recognized by human subjects in the majority of cases, we also noticed some limitations of the current method.</p>
<p>The robots used in our experiments seemed to show hardware limitations, which made it harder to generate negative expressions than positive ones. The mouth corners of both the robots cannot directly move down, making it hard for the two robots to express sadness the way humans did in the training images. Also in the case of the neutral expression on Elenoide, human subjects would sometimes mistake it for happy, due to the constantly slightly upward pull of the mouth corners. One question here is whether we require robots to show negative expressions and if yes, whether they should show these in the same way humans would or not. Besides considering full ranges for expression generation possibilities in hardware design, it might also be interesting to investigate if there would be ways to express sad on our robots, which would deviate from the human training images but could still be classified correctly by human subjects.</p>
<p>Another limitation of our method in the current form is that even though we are able to generate multiple joint configurations per expression, they do not directly map to a level of intensity of the expression. However, humans usually are able to generate and recognize different intensities in expressions and we, therefore, consider this an interesting direction for extending our method in future&#x20;work.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Conclusion and Future Work</title>
<p>We introduce a novel framework, called ExGenNet, to optimize facial expressions for the robots Alfie and Elenoide. This deep generative approach is based on neural networks for recognizing expressions. Using our ExGenNet, we obtained a range of joint configurations for Alfie to be able to express angry, happy, neutral, sad, and surprise and for Elenoide to express happy, neutral, and surprise. The limitations in the degrees of freedom in the mouth and eyebrows of Elenoide prevent ExGenNet from being able to generate negative expressions like sad and angry. In the pilot study, humans were asked to recognize the facial expressions generated by the two robots. The Pearson&#x2019;s correlation test showed that while angry, happy, neutral, and surprise are positively correlated, sad is negatively correlated in the case of Alfie. In the case of Elenoide, all three expressions surprise, happy, and neutral are positively correlated. However, the result for neutral in the case of Elenoide was not significant. In future work, one should conduct human-robot interaction (HRI) experiments where the robots are able to recognize the human facial expressions and generate their own facial expressions accordingly. One should also extend facial expressions to other modalities such as gesture-based expressions in the case of Elenoide.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://kdef.se/download-2/index.html">https://kdef.se/download-2/index.html</ext-link>.</p>
</sec>
<sec id="s7">
<title>Ethics Statement</title>
<p>Ethical review and approval was not required for the study on human participants in accordance with the local legislation and institutional requirements. The patients/participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="s8">
<title>Author Contributions</title>
<p>NR contributed by writing the manuscript, coding, planning, preparing and executing the experiments. The idea of the proposed approach came from JP. JP, DK and RS-H further contributed to development of proposed approach. DK and CT helped in planning of the experiments and writing the manuscript. RS-H, JP and KK also helped in writing the manuscript. CT helped in conducting experiments with Alfie. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>This research was funded by the German Research Foundation (DFG, Deutsche Forschungsgemeinschaft). The work was supported by the German Federal Ministry of Education and Research (BMBF) project 01IS20045 (IKIDA). The authors would like to thank the leap in time foundation for the grateful funding of the project. The research was also conducted as part of RoboTrust, a project of the Centre Responsible Digitality.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>The authors would like to thank Vignesh Prasad for his comments and insights that helped in improving this&#x20;paper.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>T. U.</given-names>
</name>
<name>
<surname>Hossain</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hossain</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>ul Islam</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Andersson</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Facial Expression Recognition Using Convolutional Neural Network with Data Augmentation</article-title>,&#x201d;. <comment>Vision Pattern Recognition (icIVPR)</comment> in <conf-name>2019 Joint 8th International Conference on Informatics, Electronics Vision (ICIEV) and 2019&#x20;3rd International Conference on Imaging</conf-name>, <fpage>336</fpage>&#x2013;<lpage>341</lpage>. <pub-id pub-id-type="doi">10.1109/ICIEV.2019.8858529</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Barros</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Weber</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wermter</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Emotional Expression Recognition with a Cross-Channel Convolutional Neural Network for Human-Robot Interaction</article-title>,&#x201d; in <conf-name>2015 IEEE-RAS 15th International Conference on Humanoid Robots (Humanoids)</conf-name>, <fpage>582</fpage>&#x2013;<lpage>587</lpage>. <pub-id pub-id-type="doi">10.1109/humanoids.2015.7363421</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bennett</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>&#x160;abanovi&#x107;</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Deriving Minimal Features for Human-like Facial Expressions in Robotic Faces</article-title>. <source>Int. J.&#x20;Soc. Robotics</source> <volume>6</volume>, <fpage>367</fpage>&#x2013;<lpage>381</lpage>. <pub-id pub-id-type="doi">10.1007/s12369-014-0237-z</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Randhavane</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Prinja</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kapsaskis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gray</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <source>The Emotionally Intelligent Robot: Improving Social Navigation in Crowded Environments</source>. <comment>
<italic>ArXiv</italic> abs/1903</comment>, <comment>, 03217</comment>. </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breazeal</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Buchsbaum</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gray</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gatenby</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Blumberg</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Learning from and about Others: Towards Using Imitation to Bootstrap the Social Understanding of Others by Robots</article-title>. <source>Artif. Life</source> <volume>11</volume>, <fpage>31</fpage>&#x2013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1162/1064546053278955</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breazeal</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Emotion and Sociable Humanoid Robots</article-title>. <source>Int. J.&#x20;human-computer Stud.</source> <volume>59</volume>, <fpage>119</fpage>&#x2013;<lpage>155</lpage>. <pub-id pub-id-type="doi">10.1016/s1071-5819(03)00018-1</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Canedo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Neves</surname>
<given-names>A. J.&#x20;R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Facial Expression Recognition Using Computer Vision: A Systematic Review</article-title>. <source>Appl. Sci.</source> <volume>9</volume>, <fpage>4678</fpage>. <pub-id pub-id-type="doi">10.3390/app9214678</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Churamani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Barros</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Strahl</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wermter</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Learning Empathy-Driven Emotion Expressions Using Affective Modulations</article-title>,&#x201d; in <conf-name>2018 International Joint Conference on Neural Networks (IJCNN)</conf-name>. <pub-id pub-id-type="doi">10.1109/ijcnn.2018.8489158</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cid</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Moreno</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bustos</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>N&#xfa;&#xf1;ez</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Muecas: A Multi-Sensor Robotic Head for Affective Human Robot Interaction and Imitation</article-title>. <source>Sensors</source> <volume>14</volume>, <fpage>7711</fpage>&#x2013;<lpage>7737</lpage>. <pub-id pub-id-type="doi">10.3390/s140507711</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>de Graaf</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Allouch</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Van Dijk</surname>
<given-names>J.&#x20;A.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Long-term Acceptance of Social Robots in Domestic Environments: Insights from a User&#x2019;s Perspective</source>. </citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ge</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2008</year>). <source>A Facial Expression Imitation System in Human Robot Interaction</source>. </citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hoffman</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Breazeal</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2006</year>). <source>Robotic Partners&#x2019; Bodies and Minds: An Embodied Approach to Fluid Human-Robot Collaboration</source>. <comment>AAAI Workshop - Technical Report</comment>. </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horii</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nagai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Asada</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Imitation of Human Expressions Based on Emotion Estimation by Mental Simulation</article-title>. <source>Paladyn, J.&#x20;Behav. Robotics</source> <volume>7</volume>. <pub-id pub-id-type="doi">10.1515/pjbr-2016-0004</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2006</year>). <source>Development of a Facial Expression Imitation System</source>, <fpage>3107</fpage>&#x2013;<lpage>3112</lpage>. </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>King</surname>
<given-names>D. E.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Dlib-ml: A Machine Learning Toolkit</article-title>. <source>J.&#x20;Machine Learn. Res.</source> <volume>10</volume>, <fpage>1755</fpage>&#x2013;<lpage>1758</lpage>. </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep Facial Expression Recognition: A Survey</article-title>. <source>IEEE Trans. Affective Comput.</source>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1109/taffc.2020.2981446</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundqvist</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Flykt</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>&#xd6;hman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>The Karolinska Directed Emotional Faces (Kdef)</article-title>. <source>CD ROM. Department Clin. Neurosci. Psychol. section, Karolinska Institutet</source> <volume>91</volume>, <fpage>2</fpage>. </citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Meghdari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shouraki</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Siamy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shariati</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <source>The Real-Time Facial Imitation by a Social Humanoid Robot</source>. </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mehrabian</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1968</year>). <article-title>Communication without Words</article-title>. <source>Psychol. Today</source> <volume>2</volume>, <fpage>53</fpage>&#x2013;<lpage>56</lpage>. </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mori</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>MacDorman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kageki</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>The Uncanny valley [from the Field]</article-title>. <source>IEEE Robot. Automat. Mag.</source> <volume>19</volume>, <fpage>98</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.1109/mra.2012.2192811</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nicolescu</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Mataric</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Learning and Interacting in Human-Robot Domains</article-title>. <source>IEEE Trans. Syst. Man. Cybern. A.</source> <volume>31</volume>, <fpage>419</fpage>&#x2013;<lpage>430</lpage>. <pub-id pub-id-type="doi">10.1109/3468.952716</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Noh</surname>
<given-names>J.-y.</given-names>
</name>
<name>
<surname>Neumann</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>Expression Cloning</article-title>,&#x201d; in <source>ACM SIGGRAPH 2006 Courses</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <comment>SIGGRAPH &#x2019;06</comment>. <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>22</fpage>. <comment>&#x2013;es</comment>. <pub-id pub-id-type="doi">10.1145/1185657.1185862</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pighin</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hecker</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lischinski</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Szeliski</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Salesin</surname>
<given-names>D. H.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>Synthesizing Realistic Facial Expressions from Photographs</article-title>,&#x201d; in <source>ACM SIGGRAPH 2006 Courses</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>19</fpage>. <comment>SIGGRAPH &#x2019;06</comment>. <comment>es</comment>. <pub-id pub-id-type="doi">10.1145/1185657.1185859</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pumarola</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Agudo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Martinez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sanfeliu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Moreno-Noguer</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Ganimation: One-Shot Anatomically Consistent Facial Animation</source>. </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rawal</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Stock-Homburg</surname>
<given-names>R. M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Facial Emotion Expressions in Human-Robot Interaction: A Survey</article-title>. <source>To appear Int. J.&#x20;Soc. Robotics</source>. </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ray</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mondada</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>What Do People Expect from Robots?</article-title> <fpage>3816</fpage>&#x2013;<lpage>3821</lpage>. </citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Silva</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Soares</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Esteves</surname>
<given-names>J.&#x20;S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Mirroring Emotion System-On-Line Synthesizing Facial Expressions on a Robot Face</article-title>,&#x201d; in <conf-name>2016 8th International Congress on Ultra Modern Telecommunications and Control Systems and Workshops (ICUMT)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>213</fpage>&#x2013;<lpage>218</lpage>. <pub-id pub-id-type="doi">10.1109/icumt.2016.7765359</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Simul</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>Ara</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Islam</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>A Support Vector Machine Approach for Real Time Vision Based Human Robot Interaction</article-title>,&#x201d; in <conf-name>2016 19th International Conference on Computer and Information Technology (ICCIT)</conf-name>, <fpage>496</fpage>&#x2013;<lpage>500</lpage>. <pub-id pub-id-type="doi">10.1109/iccitechn.2016.7860248</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>