<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1359782</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2024.1359782</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Unraveling the thread: understanding and addressing sequential failures in human-robot interaction</article-title>
<alt-title alt-title-type="left-running-head">Tisserand et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2024.1359782">10.3389/frobt.2024.1359782</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tisserand</surname>
<given-names>Lucien</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2332872/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Stephenson</surname>
<given-names>Brooke</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2759754/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Baldauf-Quilliatre</surname>
<given-names>Heike</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2620333/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lefort</surname>
<given-names>Mathieu</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Armetta</surname>
<given-names>Fr&#xe9;d&#xe9;ric</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Interactions, Corpus, Apprentissages, Repr&#xe9;sentations (ICAR) UMR5191</institution>, <institution>Centre National de la Recherche Scientifique</institution>, <institution>ENS de Lyon and Universit&#xe9; Lyon 2</institution>, <institution>Labex ASLAN</institution>, <addr-line>Lyon</addr-line>, <country>France</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>University Lyon</institution>, <institution>Universit&#xe9; Claude Bernard Lyon 1</institution>, <institution>CNRS</institution>, <institution>INSA Lyon</institution>, <institution>Laboratoire d&#x2019;InfoRmatique en Image et Syst&#xe8;mes d&#x2019;information (LIRIS) UMR5205</institution>, <addr-line>Villeurbanne</addr-line>, <country>France</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/971894/overview">Frank Foerster</ext-link>, University of Hertfordshire, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1563125/overview">Olov Engwall</ext-link>, Royal Institute of Technology, Sweden</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2645742/overview">Andreas Liesenfeld</ext-link>, Radboud University, Netherlands</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Lucien Tisserand, <email>lucien.tisserand@ens-lyon.fr</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>09</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1359782</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>08</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Tisserand, Stephenson, Baldauf-Quilliatre, Lefort and Armetta.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Tisserand, Stephenson, Baldauf-Quilliatre, Lefort and Armetta</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Interaction is a dynamic process that evolves in real time. Participants interpret and orient themselves towards turns of speech based on expectations of relevance and social/conversational norms (that have been extensively studied in the field of Conversation analysis). A true challenge to Human Robot Interaction (HRI) is to develop a system capable of understanding and adapting to the changing context, where the meaning of a turn is construed based on the turns that have come before. In this work, we identify issues arising from the inadequate handling of the sequential flow within a corpus of in-the-wild HRIs in an open-world university library setting. The insights gained from this analysis can be used to guide the design of better systems capable of handling complex situations. We finish by surveying efforts to mitigate the identified problems from a natural language processing/machine dialogue management perspective.</p>
</abstract>
<kwd-group>
<kwd>human-robot interaction</kwd>
<kwd>in-the-wild</kwd>
<kwd>conversation analysis</kwd>
<kwd>sequentiality</kwd>
<kwd>grounding</kwd>
<kwd>contextualisation</kwd>
</kwd-group>
<contract-sponsor id="cn001">LabEx ASLAN<named-content content-type="fundref-id">10.13039/501100011602</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Human-Robot Interaction</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Until recently, robot systems have largely managed spoken interaction through controlled mechanisms such as finite state machines (e.g. <xref ref-type="bibr" rid="B83">Nakano and Komatani, 2020</xref>) and rule-based systems (e.g. <xref ref-type="bibr" rid="B121">Webb, 2000</xref>). These models, designed to carry out predefined tasks, can be quite rigid and commonly result in failures when confronted with the complexity of real-world situations. Unlike in controlled laboratory experiments, in open-world scenarios, it can be difficult to predict user behaviour. Users do not necessarily have a well defined representation of the robot&#x2019;s purpose or the manner in which they should engage with it (e.g. <xref ref-type="bibr" rid="B3">Arend et al., 2017</xref>). And even when the purpose is understood, if humans &#x201c;deviate&#x201d; from the designed script/progression, for example by making two requests within a single turn or by returning to an earlier discussion thread, the state of conversation can become confused. New technologies, such as large language models (LLMs), may be able to offer increased flexibility in dealing with a larger context, a broader range of topics and varied formulations, however if their training data has only a limited amount of spontaneous spoken speech, they may struggle with the interactional elements of speech, such as turn-taking, which can differ significantly from written text or scripted media. Nor can LLMs alone incorporate instantaneous feedback (e.g., backchanneling) from the user to adapt their responses on the fly. Regardless of the underlying technology, it is important to understand the nature of speech and the underlying principles guiding real-life interaction in order to optimize system design to human behaviour. This motivates the current study which analyses HRI failures through the lens of conversation analysis (CA).</p>
<p>Previous studies based on CA have identified a major problem in the case of multi-party interaction where the meaning, timing and sequential positioning of turns addressed to the machine depend on the ongoing interaction between humans (<xref ref-type="bibr" rid="B93">Porcheron et al., 2017</xref>; <xref ref-type="bibr" rid="B3">Arend et al., 2017</xref>; <xref ref-type="bibr" rid="B96">Reeves and Porcheron, 2023</xref>; <xref ref-type="bibr" rid="B33">Gehle et al., 2015</xref>; <xref ref-type="bibr" rid="B116">Tisserand and Baldauf-Quilliatre, 2024</xref>). The issue is, these human-human interactions (from which the turns addressed to the robot depend) are complex, for this interactional complexity reflects the complexity of real-world social and cultural practices, roles, identities, etc. (<xref ref-type="bibr" rid="B75">Lund et al., 2022</xref>). Detailed studies on <italic>temporally continuous</italic> and <italic>multimodal</italic> interactions with machines also show that participants use specific practices when addressing a machine: these specific practices may index what they treat as an <italic>adequate norm</italic> for this purpose. For example, CA studies have identified the keyword formatting of turns (<xref ref-type="bibr" rid="B88">Pelikan and Broth, 2016</xref>; <xref ref-type="bibr" rid="B4">Avgustis et al., 2021</xref>) or vocal commands and unilateral departures (<xref ref-type="bibr" rid="B69">Licoppe and Rollet, 2020</xref>; <xref ref-type="bibr" rid="B119">Velkovska et al., 2020</xref>). In transposing turn-taking, similarities and differences have both been demonstrated (<xref ref-type="bibr" rid="B88">Pelikan and Broth, 2016</xref>; <xref ref-type="bibr" rid="B76">Majlesi et al., 2023</xref>). These results show that there is a need to explore in detail the practices transposed from human-human encounters and the new practices that emerge in a world where people are increasingly engaging with machines through speech and physical gestures in a sustained, temporally continuous manner.</p>
<p>Other CA-based studies have proposed models based on fine-grained analyses of human-human-interactions. For example, albeit not applied to <italic>human-robot interactions</italic>, <xref ref-type="bibr" rid="B2">Aoki et al. (2006)</xref> distinguished patterns in the conversational organization of schisming (i.e., the methodical accomplishment of splitting one conversation into sub-conversations (<xref ref-type="bibr" rid="B23">Egbert, 1997</xref>)), which could be detected to adapt the audio space in group meetings between humans. Multi-modal datasets of instructed laboratory interactions between humans (introducing themselves, playing a game, etc.) have been created with the hypothesis that they could be used for HRI (<xref ref-type="bibr" rid="B118">Tuyen et al., 2023</xref>). But only very few studies try to improve the design of a conversational agent/robot by drawing on the systematicity of the conversation analytic approach applied to the data (but see <xref ref-type="bibr" rid="B74">Lohse et al. (2009)</xref> for an example). <xref ref-type="bibr" rid="B90">Pitsch (2016)</xref> (involved in the previously cited collaboration) emphasizes the limits in <italic>formalizing</italic> interactions and thus, the need to carefully define what is worth being created in terms of <italic>formal</italic> objects that could be processed by the system. She concludes that the human capability to adapt to the machine can be leveraged by making transparent and explainable the robot&#x2019;s actions.</p>
<p>This means that improving HRI requires finding 1) which human interactional norms are indexed (i.e. a reference made explicit <italic>insitu</italic>) and how it is adapted to the situation of using a robot, 2) which human norms are not made relevant, 3) which human norms are considered problematic by users during robot interactions, <italic>e.g.,</italic> because the robot is not a person, and 4) new norms relevant for interacting with a robot (that is, when people are interacting with a machine, especially in public, what are they expected to do with regards to the design of their verbal and multimodal actions. An empirical approach has the benefit of capturing displays of people making their conducts accountable (interpretable, justifiable, that can be attributed to their agency and intention) and with regards to norms of action such as the sequential norms of adjacency pairs that we present in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>. As we will show, this approach can evidence how acting according to a norm can be treated as a problem or treated as adequate, through the study of interactional phenomena like delaying, disalignment (<xref ref-type="bibr" rid="B63">Lee and Tanaka, 2016</xref>), repairing, accounts, etc.</p>
<p>In this paper, we try to respond to these challenge by investigating an in-the-wild scenario and analyzing what can be considered failures in the interaction with a social robot, based on the sequential organization evidenced by detailed analyses. In so doing, we uncover current limitations in the robot&#x2019;s programming which should be considered in future designs. We focus principally on failures that are due to deficiencies in sequence organization. Indeed, one of the core findings of conversation analysis is the sequential organization of interaction: each action projects specific follow-up actions (indexing a normative <italic>sequence of action</italic>). The temporal unfolding is thus an important aspect when modeling interaction. We develop this dimension in the theoretical section (see <xref ref-type="sec" rid="s2-1">Section 2.1</xref>). We show how some of the failures can be the outcome of humans transposing basic sequential organisation techniques that the program can not handle.</p>
<p>In the latter part of this paper, we turn our attention to possible technical solutions to the current shortcomings we observed. We also discuss the suitability of different dialogue management systems for the handling of the sequential flow in an open environment which includes non-elicited and non-guided interaction. Descriptive approaches to dialogue management have traditionally been used to handle focused service requests as they allow for precise programming and predefined responses tailored to specific tasks or inquiries. Generative AI approaches on the other hand can offer more flexibility for dealing with unexpected subjects, but their output is more difficult to control. The coupling of descriptive oriented approaches and generative AI remains a major challenge to be addressed in the coming years and is discussed in this article.</p>
<p>We first explain some of the theoretical concepts and methodological implications (<xref ref-type="sec" rid="s2">Section 2</xref>) before presenting the analyzed data (<xref ref-type="sec" rid="s3">Section 3</xref>). We then provide an analysis of some of the failures we identified (<xref ref-type="sec" rid="s4">Section 4</xref>) and finally discuss which technical solutions could be applied (<xref ref-type="sec" rid="s5">Section 5</xref>).</p>
</sec>
<sec id="s2">
<title>2 Theoretical underpinnings and methodological implications</title>
<p>In this section, we briefly introduce the sequential organization of conversations as evidenced by <italic>Conversation Analysis</italic> (<italic>CA</italic>) following a corpus-based approach (see <xref ref-type="sec" rid="s2-1">Section 2.1</xref>). After that, we explain the methodological implications and the kind of knowledge we gain when applying <italic>CA</italic> to <italic>HRI studies</italic> that aim at improving human-robot interactions (see <xref ref-type="sec" rid="s2-2">Section 2.2</xref>). Finally, we delineate what dimensions are involved in defining a failure with regards to the sequential organization of interactions, but also, by taking into account orientations taken by <italic>HRI studies</italic> (see <xref ref-type="sec" rid="s2-3">Section 2.3</xref>).</p>
<sec id="s2-1">
<title>2.1 The sequential organization of conversations: norms, context and HRI</title>
<p>In interaction, the context is seen by <italic>Ethnomethodological Conversation Analysis</italic> (hereafter <italic>EMCA</italic>) as a resource for interpretation by means of a mechanics of intention and expectancy (<xref ref-type="bibr" rid="B65">Levinson, 2006</xref>): prototypically, a question creates the expectancy for an answer. Through <italic>sequence organization</italic> (<xref ref-type="bibr" rid="B105">Schegloff, 2007</xref>; <xref ref-type="bibr" rid="B53">Kendrick et al., 2020</xref>) humans provide their interlocutor with the opportunity to show how they have interpreted a previous production (e.g., as a question, an offer, a response, an unexpected response). Thus, for a same turn produced by Pepper such as &#x201c;How can I help you? Don&#x2019;t hesitate to ask me what I can do&#x201d;, participants can display that they interpret it as a directive (responding &#x201c;oh okay then what can you do&#x201d;) or as a proposal (responding with &#x201c;thank you but you can&#x2019;t help me&#x201d;). These expectations of a next action are verifiable by the way others adapt to them (or not) in real time: the interaction is seen as a temporally continuous and incremental process and not a purely logical and serial one.</p>
<p>Although this sequential organization is operative at different levels of granularity and in different modalities, <italic>adjacency pairs</italic> account most strikingly for the accomplishment of such normative conducts and expectancy. The <italic>sequence</italic> of <italic>adjacency pairs</italic> is the relationship between two actions that are paired as types of action and accomplished in an orderly manner by (at least) two participants contiguously (<xref ref-type="bibr" rid="B105">Schegloff, 2007</xref>). A <italic>First Pair Part</italic> (e.g., a question) makes conditionally relevant a <italic>Second Pair Part</italic> (e.g., an answer), so that a response can be &#x201c;officially absent&#x201d; (<xref ref-type="bibr" rid="B101">Schegloff, 1968</xref>) in the continuous flow of interaction.</p>
<p>In the case for <italic>human-robot interactions</italic> and more generally interactions with <italic>conversational user interfaces</italic>, it has been evidenced that users transpose parts of the <italic>adjacency pair</italic> norms such as the type of next action made conditionally relevant (<xref ref-type="bibr" rid="B88">Pelikan and Broth, 2016</xref>; <xref ref-type="bibr" rid="B97">Reeves et al., 2018</xref>; <xref ref-type="bibr" rid="B28">Fischer et al., 2019</xref>; <xref ref-type="bibr" rid="B69">Licoppe and Rollet, 2020</xref>) and therefore conversation analysis could provide specifications for a system to handle such projections. This is why the present study focuses on <italic>sequential failures</italic> where a breach in conditional relevancy happens (see <xref ref-type="sec" rid="s4">Section 4</xref>). In <italic>HRI studies</italic>, while sequencing between actions is very important for understanding the progression of dialogue, methods for modelling this phenomenon explicitly have only recently started to be investigated (<xref ref-type="bibr" rid="B22">Duran, 2023</xref>; <xref ref-type="bibr" rid="B59">Kunneman and Hindriks, 2022</xref>).</p>
</sec>
<sec id="s2-2">
<title>2.2 Methodological implications when applying conversation analysis to HRI</title>
<p>The <italic>sequential</italic> dimension of interaction is researched through the &#x201c;<italic>sequential</italic>&#x201d; <italic>analysis</italic> of transcribed data. This process can be summarized as addressing the question &#x201c;why that now&#x201d; (<xref ref-type="bibr" rid="B106">Schegloff and Sacks, 1973</xref>) each time an accountable action is produced by an interactant. This question is posed at each stage of the interaction (turn-by-turn, second by second). Paying attention to the details of the temporal unfolding of turns that are exchanged between two participants allows the researcher to evidence their relationship with norms of action that are indexed and can be distinguished (<xref ref-type="bibr" rid="B82">Muhle, 2024</xref>). When analyzing <italic>human-robot interactions</italic> with this theoretical and methodological framework, some insights can thus be provided so that the designer and the software engineer can decide on alternative choices that have an impact on the robot&#x2019;s output, with the aim of aligning to the users&#x2019; orientation towards the norm they make noticeable (<xref ref-type="bibr" rid="B89">Pelikan et al., 2024</xref>).</p>
<p>For example, below is the detailed transcription of several users&#x2019; response to the robot&#x2019;s greeting and/or offer extracted from the corpus used in the present study:</p>
<p>
<monospace>220324&#x5f;48&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;Case 1</monospace>
</p>
<p>
<monospace>01 robot:&#x2003;&#x2003;hello (&#x2e;) je peux t&#x2019;aider&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello (&#x2E;) can I help you&#x2f;</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.8)</monospace>
</p>
<p>
<monospace>03 human:&#x2003;&#x2003;ah oui</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;oh yes</monospace>
</p>
<p>
<monospace>220309&#x5f;84&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;Case 2</monospace>
</p>
<p>
<monospace>01 robot:&#x2003;&#x2003;coucou\ (&#x2e;) je peux t&#x2019;aider&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hi&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;can I help you</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.5)</monospace>
</p>
<p>
<monospace>03&#x2003;human:&#x2003;euh: oui&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;uh: yes&#x2f;</monospace>
</p>
<p>
<monospace>220929&#x5f;17A&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;Case 3</monospace>
</p>
<p>
<monospace>01 robot:&#x2003;&#x2003;hello (&#x2e;) moi c&#x2019;est pepper (&#x2e;) je peux t&#x2019;aider&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hi&#x2003;&#x2003;(&#x2e;) my name is pepper (&#x2e;) can I help you</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.9)</monospace>
</p>
<p>
<monospace>03 human:&#x2003;euh: oui&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;uh: yes&#x2f;</monospace>
</p>
<p>
<monospace>220324&#x5f;84&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;Case 4</monospace>
</p>
<p>
<monospace>01 robot:&#x2003;&#x2003;salut (&#x2e;) je peux t&#x2019;aider&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hi&#x2003;(&#x2e;) can I help you&#x2f;</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.1)</monospace>
</p>
<p>
<monospace>03&#x2003;human1:&#x2003;oui:</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;yes:</monospace>
</p>
<p>
<monospace>04&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.2)</monospace>
</p>
<p>
<monospace>05 human2:&#x2003;tu veux lui demander quoi&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;what do you wanna ask&#x2f;</monospace>
</p>
<p>
<monospace>06&#x2003;human1:&#x2003;ch&#xe9; pas</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;dunno</monospace>
</p>
<p>
<monospace>220926&#x5f;28&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;Case 5</monospace>
</p>
<p>
<monospace>01 robot:&#x2003;&#x2003;hello (&#x2e;) moi c&#x2019;est pepper (&#x2e;) je peux t&#x2019;aider&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hi &#x2003;(&#x2e;) my name is pepper (&#x2e;) can I help you</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.3)</monospace>
</p>
<p>
<monospace>03 human:&#x2003;euh: oui je: je voudrais: j&#x2019;ai besoin de ton aide</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;uh: yes I: I wou:ld &#x2003;&#x2003;I need &#x2003;&#x2003; your help</monospace>
</p>
<p>
<monospace>04 human:&#x2003;comment (0.6) peux-tu m&#x2019;aider&#x2f;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;how&#x2003;&#x2003;(0.6) can you help me&#x2f;</monospace>
</p>
<p>
<monospace>220324&#x5f;48&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;&#x2d;Case 6</monospace>
</p>
<p>
<monospace>01 robot:&#x2003;&#x2003;je peux te donner des infos ou t&#x2019;orienter\</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;I can give you informations or orient you</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.7)</monospace>
</p>
<p>
<monospace>03&#x2003;human:&#x2003;ben::j::&#x2018; veux:::&#x2003;(1.1) j&#x2018; vais prendre un caf&#xe9; s&#x2019;il vous pla&#xee;t</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;well:: I:: will:::&#x2003;(1.1) I&#x2019;ll have a coffee please</monospace>
</p>
<p>While such cases can be seen as a collection of acceptances (Cases 1&#x2013;5) and an isolated case of request (Case 6), they can also be seen as a collection of the normative adjacency pair [offer<inline-formula id="inf1">
<mml:math id="m1">
<mml:mo>&#x2192;</mml:mo>
</mml:math>
</inline-formula>acceptance<inline-formula id="inf2">
<mml:math id="m2">
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:math>
</inline-formula>reject<inline-formula id="inf3">
<mml:math id="m3">
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:math>
</inline-formula>request] being indexed as problematic and not fully applicable, thanks to the detailed account of events that occurred. Indeed, the gaps (Cases 1&#x2013;4 and 6), the hesitation markers (Cases 2&#x2013;3 and 5&#x2013;6) and the voice lenghtenings (Cases 2&#x2013;6) show that the humans do not treat the robot&#x2019;s offer as preferably projecting first and foremost an acceptance or a direct request (<xref ref-type="bibr" rid="B92">Pomerantz and Heritage, 2012</xref>; <xref ref-type="bibr" rid="B105">Schegloff, 2007</xref>). More drastically, in Case 1, the <italic>change-of-state token</italic> (<xref ref-type="bibr" rid="B41">Heritage, 1984</xref>) even displays that having to participate in such a sequence is surprising. Finally, other accounts also inform us about the users&#x2019; relationship to the offer sequence (<xref ref-type="bibr" rid="B104">Schegloff, 2002</xref>; <xref ref-type="bibr" rid="B43">Heritage and Watson, 1979</xref>). The account asked by human2 to human1, who accepted the offer (Lines 5-6 of Case 4), but also, the return question that initiates the repair of the meaning of a previous acceptance (Line 4 of Case 5) show that this action&#x2013;the acceptance of the initial generic offer&#x2013;is not treated with the same social implications as the actual actions accomplished through the offer sequence as a resource (<xref ref-type="bibr" rid="B54">Kendrick and Drew, 2014a</xref>). If we build a collection of cases that all include one of these details (especially the resources that delay the acceptance), then such a collection can quickly grow as such patterns are common and not surprising in this situation where average users do not know the purpose of the robot.</p>
<p>When these results are oriented towards HRI, drawing attention to the fact that these generic initial offers appear as inappropriate can then lead the designer to change the scenario, for example, by completely removing this robot action at this moment, and instead, proposing an alternative (i.e. the scenario does not require that the user actually needs something), if one wants to align to the norms that the users orient to or not. Or, having shown that phenomena such as voice lengthening, delayed answers or hesitation markers regularly happen in a localized context (this generic initial offer can be accounted as inappropriate by the humans), one can also decide to build a system that can specifically handle such localized phenomena (see <xref ref-type="sec" rid="s5-2">Section 5.2</xref>) for these can be responsible for sequential failures (as shown in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>). Furthermore, the resources identified can be treated as cues, systematically annotated, and queried in the corpus in order to identify other contexts of inappropriate expectancy. What matters is to make an explicit and justified decision with regards to such results, as we explain in <xref ref-type="sec" rid="s4">Section 4</xref>.</p>
<p>Between a human and a robot in a public space, the presence of other humans (hypothetical or verified) raises, for the users, the practical issue of producing conducts that are also adequate from the point of view of other humans. Thus, in such situations, specific practices for exchanging turns with this machine are made relevant, monitored and aligned on as a norm oriented to by the users. <italic>HRI studies</italic> are mainly based on experimentation in laboratory settings. This type of setting biases the orientation towards norms (<xref ref-type="bibr" rid="B90">Pitsch, 2016</xref>). When users aim to complete a specified script or activity within experimental boundaries, they also align with the experimenter&#x2019;s expectations. This means they might persist through the robot&#x2019;s failures, rather than solely responding to the interaction&#x2019;s emergent goals with the robot. We therefore claim that research based on real-world situations is crucial to improve the design of user interfaces based on conversational dynamics and social robots tailored to it. Naturally occurring interactions or in-the-wild experiments (where users can act&#x2013;or not&#x2013;with the agent the way they want) are necessary to understand what is treated by humans as failures with regards to norms in the interaction that can only be relevant in such a context.</p>
<p>With regard to the relationship to human-human interactional norms when interacting with a Pepper robot in the public space, it has been shown elsewhere that users account for a deviation when transposing other social norms of interaction. For example (<xref ref-type="bibr" rid="B69">Licoppe and Rollet, 2020</xref>, p.172) evidenced that when the robot responds appropriately and is successful in achieving basic sequences of action, the fact that this performance can be assessed as surprising or pleasant by the humans not only deviates from a human-human normal and subliminal expectancy (<xref ref-type="bibr" rid="B27">Enfield and Sidnell, 2021</xref>), but it also indexes the fact that the robot is first and foremost treated as not being a competent interactant. When it is the case for the humans to respond appropriately, it has been shown that they also account for a deviation by the means of systematically treating as humorous the fact that they have to treat the offer as implying <italic>preference organization</italic> (i.e., acceptance are more straightforwardly produced than rejection that are more elaborated), when producing the <italic>offer reject</italic> (<xref ref-type="bibr" rid="B116">Tisserand and Baldauf-Quilliatre, 2024</xref>).</p>
</sec>
<sec id="s2-3">
<title>2.3 The collaborative definition of a failure</title>
<p>Previous epistemological reflections on research designs bridging <italic>EM(CA)</italic> and <italic>system design</italic> (<xref ref-type="bibr" rid="B10">Button, 1990</xref>; <xref ref-type="bibr" rid="B11">Button and Sharrock, 1995</xref>; <xref ref-type="bibr" rid="B17">Dourish and Button, 1998</xref>; <xref ref-type="bibr" rid="B16">Dourish, 2006</xref>; <xref ref-type="bibr" rid="B90">Pitsch, 2016</xref>; <xref ref-type="bibr" rid="B89">Pelikan et al., 2024</xref>) have emphasized the need to make explicit the rationalization used when CA results are applied to system design (e.g., in the form of <italic>design implications</italic> or so-called <italic>cues</italic>) and for what situated practical purposes. When addressing such potential collaborations, useful observations have been drawn with regards to what it really means to attempt to model human interaction by transforming into rules the norms identified by CA. Firstly, a rule-based system cannot be compared with human interaction (<xref ref-type="bibr" rid="B10">Button, 1990</xref>; <xref ref-type="bibr" rid="B11">Button and Sharrock, 1995</xref>), nor can a statistics-based speech system (<xref ref-type="bibr" rid="B102">Schegloff, 1996</xref>, p.22). Indeed, a human may expect from a machine a set of appropriate answers, that is, answers that can only comply with a particular expectancy. From another human, one may always treat an answer as <italic>accountable</italic>, that is, the fact that the respondent had a reason to do so in this way (<xref ref-type="bibr" rid="B27">Enfield and Sidnell, 2021</xref>). Secondly, when &#x201c;indexical expressions&#x201d; (indexicality is the essential property of human accountable conducts whose meaning can always be negotiated through the mechanics explained in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>) are transformed into &#x201c;ideal expressions&#x201d; (with a self-contained definite meaning/purpose), they participate in a structure missing this accountability (<xref ref-type="bibr" rid="B32">Garfinkel and Sacks, 1970</xref>, p.339). As these observations were oriented towards goals, in the present study, we draw on the paradigms they propose in order to redefine our goal, which is the enhancement of &#x201c;facilitating&#x201d; human-robot interaction rather than &#x201c;reproducing&#x201d; naturally-occurring conversation.</p>
<p>Thus the enhancement is explicitly measured through variables that do not account for the machine&#x2019;s conversational competence in a natural sense. In our case, <italic>cues</italic> are identified in the same vein as &#x201c;ideal expressions&#x201d; that can &#x201c;enhance&#x201d; the human-robot interaction from an <italic>HRI</italic> perspective. Our rationale for envisioning such enhancement is the fact that the specific practices identified by CA can be seen as the users&#x2019; orientation towards a rule-based system (the timing in turn-taking management, keyword formatting, expecting that the agent has only one possible answer, etc.), and this approach is in line with <italic>HRI studies</italic> which demonstrated that the users&#x2019; acceptance of a robot may not be based on human resemblance (<xref ref-type="bibr" rid="B34">Ghosh and Ghosh, 2021</xref>; <xref ref-type="bibr" rid="B85">Nazir et al., 2023</xref>).</p>
<p>Drawing on these previous work, in this paper, a &#x201c;failure&#x201d; is seen as the intersection of four dimensions that are made explicit:<list list-type="simple">
<list-item>
<p>1. What the humans and the robot do with regards to interactional norms: whether they transpose, problematize, or orient towards a machine adequacy, as explained in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>,</p>
</list-item>
<list-item>
<p>2. If and how the participants treat what happened in the first dimension as a failure or a problem for the ongoing interaction,</p>
</list-item>
<list-item>
<p>3. What the system did,</p>
</list-item>
<list-item>
<p>4. The fact that methods exist in order to handle the issues identified in (1) and (3), that is, the failure is &#x201c;computable&#x201d; (or it can be handled by design practices).</p>
</list-item>
</list>
</p>
<p>A sequential failure occurs in the first dimension when conditional relevancy is breached between the human and the robot and when the sequential organization of conversation is involved in such a failure. Conditional relevancy is breached when:<list list-type="simple">
<list-item>
<p>1. The robot produces an unexpected type of response or no response at all when it should</p>
</list-item>
<list-item>
<p>2. The robot takes the turn when it should not or</p>
</list-item>
<list-item>
<p>3. The user does not understand that it was his/her turn and that an action was expected.</p>
</list-item>
</list>
</p>
<p>Note that the absence of a response followed by a self-repair from the robot (such as &#x201c;sorry I did not understand&#x201d;) is not a sequential failure as it accounts for the missing response and potentially initiates a repair from the human. Here, the role of Conversation Analysis is to provide the analysis of the first and the second dimensions, thereby highlighting what type of norm-related problem exist. Thus it may justify the importance of success of an <italic>HRI</italic> approach that solved a particular problem that can be handled. However, the humans may treat merely as a failure (second dimension) what <italic>HRI</italic> can identify as a problem that can be resolved.</p>
<p>Furthermore, the second dimension might help categorizing the events as explicitly not a failure at all by analyzing the users&#x2019; orientation towards purposely putting the robot into a failure situation, which is a pervasive situation that we might term (albeit ironically) a &#x201c;successful failure&#x201d; from the user perspective. This perspective can be useful, outside the laboratory, when such a failure can not be handled by state of the art <italic>HRI</italic> and <italic>NLP</italic> solutions. We present these four dimensions to emphasize the fact that choices are made according to the perspective (user or designer) adopted (<xref ref-type="bibr" rid="B89">Pelikan et al., 2024</xref>).</p>
</sec>
</sec>
<sec id="s3">
<title>3 The data</title>
<p>We decided to investigate a service-encounter setting since it provides a rather simple and strongly normative, asymmetrical sequence organization (asymmetry of roles, turn allocation, needs, lexical choice, etc., see <xref ref-type="bibr" rid="B18">Drew, 1992</xref>; <xref ref-type="bibr" rid="B42">Heritage, 1998</xref>), where we can expect humans to draw on everyday sequential mechanics of conditional relevance (e.g. requests/offers and its acceptance/refusal, see <xref ref-type="bibr" rid="B19">Drew and Couper-Kuhlen, 2014</xref>). But also, it is a plausible purpose for a commercial robot like Pepper, and this attention paid to the authenticity of commercial robots used in public places explains why its initial programming is based on its built-in functionalities only. The situation has been designed so as to reproduce the typical way an institution (in this case, a university library) showcases a robot (it should serve a purpose, the services that it provides must be doable by a human, it must not disturb the environment). This setting allowed us to test if users spontaneously treat the robot&#x2019;s turns (verbal turns as well as bodily orientations) as actions demonstrating participation in a regular desk service encounter that they have to adapt to the robot, without further guidance, contrary to settings in a public space where robots greet the users and propose an activity with a limited set of answers (e.g. <xref ref-type="bibr" rid="B6">Ben-Youssef et al., 2017</xref>; <xref ref-type="bibr" rid="B33">Gehle et al., 2015</xref>). In the present study, users were passersby, users of the library, that had not been recruited beforehand.</p>
<p>We chose to use a descriptive model for this study to cover the scope of possible domain-specific information and actions requested by the users. The program (QiSDK<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref>) recognizes keywords that trigger a state machine to select a state specified on the diagram of transition states (this is performed through a matching function). This design choice is subject to the inherent limitations in descriptive approaches, but allows us to keep the application under control and prevent problems coming from more black box approaches (e.g., AI generative models). While this model may not represent the current state of the art, it is sufficient to produce analyzable data and to later extend the model, bearing in mind that each of the two approaches (descriptive-based versus generative-based) has its own specific drawbacks (discussed in <xref ref-type="sec" rid="s5-1">Section 5.1</xref>). Moreover, the failures we will discuss go beyond the limits of the descriptive model used, as it is the human recurrent practices that we focus on.</p>
<p>The robot was placed in the vicinity of the reception desk, at the entrance of the university library (<xref ref-type="fig" rid="F1">Figure 1</xref>). As the robot was not programmed to move (only to rotate), it was easier to define a recording area. Two large angle cameras were placed in order to capture the whole scene and especially to understand how users approached the robot before the opening of the interaction since it might be important for its unfolding. We also recorded the audio and video streams from Pepper&#x2019;s tablet.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>An image from one of the large angle cameras, demonstrating the placement of Pepper in the library entrance where people are passing by. The reception desk is 3 m away on the left, outside of the frame.</p>
</caption>
<graphic xlink:href="frobt-11-1359782-g001.tif"/>
</fig>
<p>With regards to personal data protection, posters were placed near the various entrances of the library. After each interaction, a team member obtained signed consent, otherwise the data was deleted. Eleven recording sessions took place, leading to approximately 9 h of human-robot interactions (N &#x3d; 730). A subset of them, those where consent was specifically given for online sharing for research purpose, will be made available in the future.</p>
<p>The data have been transcribed, time-aligned and annotated with regards to adjacency pair norms and repair practices (<xref ref-type="bibr" rid="B115">Tisserand et al., 2023</xref>) in ELAN<xref ref-type="fn" rid="fn2">
<sup>2</sup>
</xref> according to ICOR<xref ref-type="fn" rid="fn3">
<sup>3</sup>
</xref> conventions. The analysis allows us to identify the following (typical) failures that we describe in the subsequent section.</p>
</sec>
<sec id="s4">
<title>4 Sequential analysis of failures</title>
<p>Our analysis here highlights four typical situations which are difficult to address for conversational agents:<list list-type="simple">
<list-item>
<p>1. When a human&#x2019;s embodied turn refers to a pervasive social practice (<xref ref-type="sec" rid="s4-1">Section 4.1</xref>),</p>
</list-item>
<list-item>
<p>2. When the human takes back the sequence initiative (<xref ref-type="sec" rid="s4-2">Section 4.2</xref>),</p>
</list-item>
<list-item>
<p>3. When the construction of a turn is not straightforward (<xref ref-type="sec" rid="s4-3">Section 4.3</xref>),</p>
</list-item>
<list-item>
<p>4. When the user orients towards multiple actions at the same time (<xref ref-type="sec" rid="s4-4">Section 4.4</xref>).</p>
</list-item>
</list>
</p>
<p>For each type, we will first present transcripts coming from the corpus. We will then explain what are the sequential moves that humans orient to in order to conduct their interaction with the robot (first dimension) and how these led to the sequential failure. We point out what caused the failure in the current dialogue system and identify the types of cues that could be used in order to prevent such failures. Modelling techniques that can be used to improve the handling of these cases are then reviewed and discussed. These approaches seek to introduce greater flexibility into dialogue systems that need to be able to handle turns in interaction with turn-taking cues, multiple actions/intents and parallel threads. Thus, the conversation analytic part is completed in <xref ref-type="sec" rid="s5">Section 5</xref> by dialogue modelling proposals from the state of the art to address the drawbacks.</p>
<sec id="s4-1">
<title>4.1 Multimodality and sequentiality with reference to ordinary activities</title>
<p>Here, the failure is the fact that the greeting turn produced by Pepper orients towards the recognition of &#x201c;opening an interaction&#x201d; while the greeting accomplished by the human indexes another type of &#x201c;greetings alone&#x201d; activity. Thus, at the end of Line 3, Pepper is accountable for making a response conditionally relevant (responding to its offer) while the human prevented this possibility in the first place:</p>
<p>
<monospace>220929_10</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;((human is passing by))</monospace>
</p>
<p>
<monospace>01&#x3e;human: &#x2003;&#x2003;salut Pepper (0.2) bonne journ&#xe9;e/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello Pepper (0.2) have a nice day/</monospace>
</p>
<p>
<monospace>02&#x3e;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.8) ((human is leaving))</monospace>
</p>
<p>
<monospace>03 robot:&#x2003;&#x2003;&#x2003;hello (0.2) moi c&#x2019;est Pepper (0.2) j&#x2018; peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello (0.2) my name is Pepper (0.2) can I help you/</monospace>
</p>
<p>Here, the robot (Line 3) produces a greeting turn orienting towards the continuation of the interaction (greeting &#x2b; self-presentation &#x2b; offer) instead of a &#x201c;greetings alone&#x201d; interaction. Activity-wise, the apparent opposition (&#x201c;hello&#x201d; vs. &#x201c;goodbye&#x201d;) (Line 1) is actually the accomplishment of an ordinary practice among humans. In settings such as workplaces, where people routinely see each other for the first time of the day, members perform &#x201c;greetings only&#x201d; interactions without stopping to walk (<xref ref-type="bibr" rid="B117">Tunser, 2014</xref>, pp. 121&#x2013;126). This is what the human is trying to perform with Pepper in the example above. Categorization-wise, he treats the robot as being part of his routine at the university library, which is also indexed by the address term &#x201c;Pepper&#x201d; packed with the greeting.<xref ref-type="fn" rid="fn4">
<sup>4</sup>
</xref>
</p>
<p>With regard to the <italic>FSM</italic> programming, the failure is that it recognized the &#x201c;hello&#x201d; (&#x201c;salut&#x201d;) greeting keyword only. However, as shown in <xref ref-type="sec" rid="s5-3">Section 5.3</xref>, multiple actions in turns are recurrent, this specific case of &#x201c;greetings alone&#x201d; is pervasive when a robot is placed in a high-traffic public space (see additional cases). Typically, this kind of failure cannot be reproduced in a laboratory setting. It is not a failure from the point of view of the user who continues on his way<xref ref-type="fn" rid="fn5">
<sup>5</sup>
</xref>, however, it is a failure in this setting as the robot has been activated and is waiting for a next turn: thus problems may arise with other passersby in the immediate future. Ideally, the robot would have produced a reciprocal terminal (e.g. aligning on the same words &#x201c;have a nice day&#x201d; <xref ref-type="bibr" rid="B105">Schegloff, 2007</xref>, pp.195&#x2013;207) or greeting exchange. Still adequately, the robot could also not respond at all, as the user, who is leaving, looses the opportunity to treat the absence of response as a breach in conditional relevancy.</p>
<p>As a workaround to the unavailable meaning of this turn as participating in a social practice (distinguishing two activities), cues are made available by the human so that these can be used by a machine in order to overcome such sequential failure:<list list-type="simple">
<list-item>
<p>&#x2022; The action-types [greetings &#x2b; terminal exchanges] in the same turn (that can be handled by multi-threaded approaches, see <xref ref-type="sec" rid="s5-4">Section 5.4</xref>);</p>
</list-item>
<list-item>
<p>&#x2022; The humans produces this turn while continuing their walk (that can be handled by &#x201c;non-verbal&#x201d; cues approach/exit, see <xref ref-type="sec" rid="s5-5">Section 5.5</xref>).</p>
</list-item>
</list>
</p>
<p>The first case that we presented emphasized the &#x201c;greeting alone&#x201d; activity by the two-action formatting of the human&#x2019;s turn and the robot&#x2019;s response. The more common cases depend on the visual cue: humans stay oriented towards continuing their walk through a torque of their body (<xref ref-type="bibr" rid="B103">Schegloff, 1998</xref>), that is, the lower part of the body is not oriented towards the robot while the upper part is. Below are variant cases of humans continuing their walk while producing a greeting, which are prone to the failure identified and can be processed through visual cues:</p>
<p>
<monospace>220318_16----------------------</monospace>
</p>
<p>
<monospace>((two humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01 human1:&#x2003;(inaud.) ((slows down in front of Pepper and waves))</monospace>
</p>
<p>
<monospace>02 robot&#x2003;:&#x2003;hello\</monospace>
</p>
<p>
<monospace>((humans laugh and leave, Pepper waits for an answer))</monospace>
</p>
<p>
<monospace>220324_15----------------------</monospace>
</p>
<p>
<monospace>((four humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01 human1:&#x2003;wesh ((waving))</monospace>
</p>
<p>
<monospace>((humans continue their walk))</monospace>
</p>
<p>
<monospace>220325_63----------------------</monospace>
</p>
<p>
<monospace>((two humans are walking towards exit))</monospace>
</p>
<p>
<monospace>01 human2:&#x2003;((approach PEP))</monospace>
</p>
<p>
<monospace>02 human2:&#x2003;sele:m xx ((waving and continuing his walk))</monospace>
</p>
<p>
<monospace>03&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(5.7)&#x2003;((human2 stops and torques towards robot))</monospace>
</p>
<p>
<monospace>04 human1:&#x2003;wesh negro/ ((waving))</monospace>
</p>
<p>
<monospace>05&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.6)&#x2003;((human leaves and laughs))</monospace>
</p>
<p>
<monospace>06 robot:&#x2003;&#x2003;salut (.)&#x2003;je peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello (.) can I help you/</monospace>
</p>
<p>
<monospace>((humans continue their walk, Pepper waits for an answer))</monospace>
</p>
<p>
<monospace>220321_21----------------------</monospace>
</p>
<p>
<monospace>((two humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01 human1: &#x2003;bonjou:r ((towards Pepper and continuing her walk))</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello:</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.7)</monospace>
</p>
<p>
<monospace>02&#x2003;robot:&#x2003;&#x2003;&#x2003;oui/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;yes</monospace>
</p>
<p>
<monospace>((humans continue their walk, Pepper wait for an answer))</monospace>
</p>
<p>
<monospace>220328_06A---------------------</monospace>
</p>
<p>
<monospace>((human is walking towards exit))</monospace>
</p>
<p>
<monospace>01&#x2003;human:&#x2003;(2.2)&#x2003;((slowly stops her walk and stayed torqued towards Pepper))</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.0)</monospace>
</p>
<p>
<monospace>03&#x2003;human:&#x2003;(1.6)&#x2003;((one step forward while staying torqued towards Pepper))</monospace>
</p>
<p>
<monospace>04&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.5)</monospace>
</p>
<p>
<monospace>05&#x2003;robot:&#x2003;&#x2003;salut (.) je peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello (.) can I help you/</monospace>
</p>
<p>
<monospace>06&#x2003;human:&#x2003;((laugh and leave))</monospace>
</p>
<p>
<monospace>220329_48----------------------</monospace>
</p>
<p>
<monospace>((three humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01&#x2003;human1:&#x2003;bonjou:r ((towards Pepper))</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello:</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.6)&#x2003;((Pepper raises head while humans continue their walk))</monospace>
</p>
<p>
<monospace>03&#x2003;human2:&#x2003;bonjou:r&#x2003;((towards Pepper))</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello:</monospace>
</p>
<p>
<monospace>04&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.4) ((humans continue their walk))</monospace>
</p>
<p>
<monospace>05&#x2003;human2:&#x2003;c&#x2019;est marrant/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;that&#x2019;s funny/</monospace>
</p>
<p>
<monospace>((humans continue their walk))</monospace>
</p>
<p>
<monospace>220329_80----------------------</monospace>
</p>
<p>
<monospace>((four humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01&#x2003;humans:&#x2003;((talking together))</monospace>
</p>
<p>
<monospace>02&#x2003;human1:&#x2003;(1.4) ((slows down and wave at Pepper))</monospace>
</p>
<p>
<monospace>((humans continue their walk))</monospace>
</p>
<p>
<monospace>220331_03----------------------</monospace>
</p>
<p>
<monospace>((two humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01&#x2003;human1:&#x2003;ah: (0.4) mec</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;oh:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;boy</monospace>
</p>
<p>
<monospace>02&#x2003;human2:&#x2003;bonjour ((towards Pepper))</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello</monospace>
</p>
<p>
<monospace>03&#x2003;human1:&#x2003;bonjour ((towards Pepper))</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello</monospace>
</p>
<p>
<monospace>((humans continue their walk))</monospace>
</p>
<p>
<monospace>220331_32----------------------</monospace>
</p>
<p>
<monospace>((four humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01&#x2003;human1:&#x2003;bonjou:r</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello:</monospace>
</p>
<p>
<monospace>02&#x2003;human2:&#x2003;((laughter))</monospace>
</p>
<p>
<monospace>((humans continue their walk))</monospace>
</p>
<p>
<monospace>220926_38B---------------------</monospace>
</p>
<p>
<monospace>((two humans are walking across the corridor))</monospace>
</p>
<p>
<monospace>01&#x2003;human1:&#x2003;wesh mon (gros) (0.3) bien ou quoi/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hey my boy&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;doing good or what</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.6) ((humans continue their walk))</monospace>
</p>
<p>
<monospace>03&#x2003;robot:&#x2003;&#x2003;tu cherches quelque chose/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;you looking for something</monospace>
</p>
<p>
<monospace>04&#x2003;human2:&#x2003;((laughter))</monospace>
</p>
<p>
<monospace>05&#x2003;human1:&#x2003;non (.) j&#x2018; te dis bonjour juste ((torqued towards Pepper))</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;no&#x2003;I&#x2019;m just saying hello</monospace>
</p>
<p>
<monospace>06&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.9) ((humans continue their walk))</monospace>
</p>
<p>
<monospace>07&#x2003;human1:&#x2003;i&#x2018; m&#x2019;a pas r&#xe9;pondu pepper\</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;it didn&#x2019;t answer pepper</monospace>
</p>
<p>
<monospace>220929_12---------------------</monospace>
</p>
<p>
<monospace>((human is walking across the corridor))</monospace>
</p>
<p>
<monospace>01&#x2003;human1:&#x2003;hi: ((towards Pepper and continuing his walk))</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.8) ((human continue his walk))</monospace>
</p>
<p>
<monospace>02&#x2003;robot :&#x2003;salut (.) moi c&#x2019;est pepper (.) j&#x2018; peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hi my name is pepper can I help you</monospace>
</p>
<p>
<monospace>((human continue his walk))</monospace>
</p>
</sec>
<sec id="s4-2">
<title>4.2 A response coupled with an initiation: When the user takes back the initiative</title>
<p>Another typical case that leads to failures when the system allows only one decision (one change of state) is the case where the user responds to the current state expectation as a second pair part of a sequence of actions and then, in the same turn, the user initiates a new sequence. Inevitably, if the robot succeeds in parsing the second pair part placed in first position and the next state is initiated with a turn, this will lead to an overlap. This is caused by a next action that does not take into account the full turn produced by the human, such as in the two cases below:</p>
<p>
<monospace>220317_05</monospace>
</p>
<p>
<monospace>01&#x2003;robot:&#x2003;je peux te donner des directions ou des informations sur la biblioth&#xe9;que\</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;I can give you directions or information about the library\</monospace>
</p>
<p>
<monospace>02&#x2003;robot:&#x2003;(.) &#xe7;a t&#x2019;int&#xe9;resse/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;(.) are you interested/</monospace>
</p>
<p>
<monospace>03&#x3e;human:&#x2003;oui/ (.) je veux un [caf&#xe9;]</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;yes/ (.) I&#x2019;d like a [coffee</monospace>
</p>
<p>
<monospace>04&#x2003;robot:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[okay\ comment est-ce que je peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;okay\ how can I help you/</monospace>
</p>
<p>
<monospace>05&#x2003;human:&#x2003;o&#xf9; sont les caf&#xe9;s/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;where is the coffee/</monospace>
</p>
<p>
<monospace>220317_29</monospace>
</p>
<p>
<monospace>01&#x2003;human:&#x2003;&#xe0; quoi tu sers/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;what are you for/</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;(0.8)</monospace>
</p>
<p>
<monospace>03&#x2003;robot:&#x2003;je peux te donner des directions ou des informations sur la biblioth&#xe9;que\</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;I can give you directions or information about the library\</monospace>
</p>
<p>
<monospace>04&#x2003;robot:&#x2003;(.) &#xe7;a t&#x2019;int&#xe9;resse/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;(.) are you interested/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.3)</monospace>
</p>
<p>
<monospace>05&#x3e;human:&#x2003;oui (0.5) [o&#xf9; sont les toilettes/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;yes (0.5) [where is the bathroom/</monospace>
</p>
<p>
<monospace>06&#x2003;robot:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[okay\ (.) comment est-ce que je peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[okay\ (.) how can I help you/</monospace>
</p>
<p>
<monospace>07&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.5)</monospace>
</p>
<p>
<monospace>08&#x2003;human:&#x2003;o&#xf9; est la machine &#xe0; caf&#xe9;/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;where is the coffee machine/</monospace>
</p>
<p>In these two cases, the program processed the human input &#x201c;yes&#x201d; and provides a response to this case of offer acceptance alone, instead of treating the succession of the two actions [acceptance &#x2b; request]. In a service encounter scenario, the response to the initial offer &#x201c;can I help you&#x201d; is prone to such situations as the initial, generic offer is projecting the immediate formulation of a request (<xref ref-type="bibr" rid="B54">Kendrick and Drew, 2014a</xref>) in such an institutional setting (<xref ref-type="bibr" rid="B71">Lindstr&#xf6;m, 2005</xref>, p.213). This offer can also be verbally accepted as the first part of the turn and hereafter, in the second part of the turn, the expected request is formulated.</p>
<p>With regard to the goal-oriented task, the difficulty here is the fact that both cases (acceptance alone, or acceptance &#x2b; follow-up action) must be discriminated between at this place. In 190 interactions that have been systematically annotated with regards to adjacency pairs, 81 acceptances have been identified, of which 54% (N &#x3d; 44) appear in a turn with a follow-up action, while 46% (N &#x3d; 37) are produced alone. Among the turns produced with a follow-up action by the human, 17 are produced with no pause or a micro-pause (12 questions/requests, 2 directives, 1 account, 1 self-identification), and 27 are produced with a pause of average 0.7 s (19 questions/requests, 6 directives, 2 accounts, 1 howareyou-sequence).</p>
<p>As already pointed out by (<xref ref-type="bibr" rid="B109">Skantze, 2021</xref>, p.13), a silence <italic>ad hoc</italic> rule would not be a solution. Here we provide additional reasons for this: as we can see in Line 5 of excerpt <monospace>220317_29</monospace>, a silence threshold would need to be longer than (0.5) seconds, which is a long time from a conversational norm standpoint. Especially, when a human responds with &#x201c;yes&#x201d;, the turn produced by Pepper can therefore be retrospectively treated as a (pre-) proposal from the robot, the &#x201c;yes&#x201d; being a <italic>go-ahead</italic> addressed to the robot, expecting an immediate new turn from it (a proposal).</p>
<p>Here, no specific verbal cue can be detected to prevent this failure. As such, with respect to norms of adjacency pairs, it is important to identify such situations. However, this overlap-proffering situation shows how important it is to have a system that is able to process overlap and/or a system that can continue to listen to possibly future talk, which Pepper&#x2019;s built-in software does not (also evidenced by <xref ref-type="bibr" rid="B88">Pelikan and Broth, 2016</xref>). Overlap in conversation can be treated ordinarily (<xref ref-type="bibr" rid="B104">Schegloff, 2002</xref>; <xref ref-type="bibr" rid="B48">Jefferson, 2004</xref>). In HRI they can be managed casually, but they can also be addressed more critically when they result of breaches in normative expectation (<xref ref-type="bibr" rid="B76">Majlesi et al., 2023</xref>). Here, from the point of view of the humans, a conventional repair practice tailored for HRI (like word selection, see <xref ref-type="bibr" rid="B113">Stommel et al., 2022</xref>) is produced in order to manage the failure.</p>
</sec>
<sec id="s4-3">
<title>4.3 Incomplete turns with turn-holding device and repairs</title>
<p>The cases below are more commonly studied than those in the previous sections (e.g. <xref ref-type="bibr" rid="B109">Skantze, 2021</xref>; <xref ref-type="bibr" rid="B5">Baumann et al., 2017</xref>). These are turns produced by humans with turn-holding devices (Line 6 in the first excerpt and Line 4 in the second excerpt below). The observable sequential failure is the overlap between the human and the robot&#x2019;s turn: that is, the robot should not respond when it does (moreover the robot&#x2019;s turn is inappropriate). This failure stems from sequential organization as in each case, the human displayed that the turn was taken and that the action would be completed. They thereby provided the interactional work of displaying their alignment on conditional relevancy and the robot did not take this into account.</p>
<p>
<monospace>220324_79</monospace>
</p>
<p>
<monospace>01&#x2003;human1:&#x2003;[oula (.) i&#x2018; m&#x2019;a vu</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[wow (.) it saw me</monospace>
</p>
<p>
<monospace>02&#x2003;robot:&#x2003;&#x2003;[ouais\ (0.2) je peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[yeah\ (0.2) can I help you/</monospace>
</p>
<p>
<monospace>03&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.3)</monospace>
</p>
<p>
<monospace>04&#x2003;human2:&#x2003;((laugh)) (0.8) ((laugh))</monospace>
</p>
<p>
<monospace>05&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.4)</monospace>
</p>
<p>
<monospace>06&#x2003;human1:&#x2003;euh:::&#x2003;(0.8) c&#x2019;est o&#xf9;/ la::: &#x2003;&#x2003; (0.8)</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hu:::m&#x2003;(0.8) where is/ the::: (0.8)</monospace>
</p>
<p>
<monospace>07&#x2003;robot:&#x2003;&#x2003;je peux t&#x2019;orienter vers diff&#xe9;rents [endroits de la biblioth&#xe9;que\</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;I can orient you towards different [places in the library\</monospace>
</p>
<p>
<monospace>08&#x2003;human1:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[((laugh))</monospace>
</p>
<p>
<monospace>220928_56</monospace>
</p>
<p>
<monospace>01&#x2003;robot:&#x2003;&#x2003;tu cherches quelque chose/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;are you looking for something/</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.2)</monospace>
</p>
<p>
<monospace>03&#x2003;human1:&#x2003;oui</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;yes</monospace>
</p>
<p>
<monospace>04&#x2003;human1:&#x2003;je cherche les euh: (0.3) le rayon [informatique</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;I&#x2019;m looking for uh: (0.3) the section [informatics</monospace>
</p>
<p>
<monospace>05&#x2003;robot:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[tu as dit que tu voulais aller o&#xf9;/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[where did you say you wanted to go/</monospace>
</p>
<p>
<monospace>06&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.5)</monospace>
</p>
<p>
<monospace>07&#x2003;human1:&#x2003;le rayon &#x2191;informatique</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;the section &#x2191;informatics</monospace>
</p>
<p>From the point of view of the state machine, each time, what happened is the program found a match for the <monospace>repair</monospace> list of words: it recognized &#x201c;where is&#x201d; (Line 6 in first excerpt) and &#x201c;I&#x2019;m looking&#x201d; (Line 4 in second excerpt).<xref ref-type="fn" rid="fn6">
<sup>6</sup>
</xref> However, the problem can be generalized as having the robot processing as complete what is, in fact, an incomplete turn designed as such. What the human is doing here is actually following the norms of conditional relevancy given the <italic>First Pair Part</italic> produced by the robot: he acknowledges the fact that a request is expected (with the &#x201c;yes&#x201d; Line 3 in excerpt <monospace>220928_56</monospace>), and that it&#x2019;s his turn (that is taken with the turn-initial hu:m in excerpt <italic>220,324_79</italic>.<xref ref-type="fn" rid="fn7">
<sup>7</sup>
</xref>
</p>
<p>Besides the user&#x2019;s display of alignment, turn-holding devices are also used within-turn: these are the filler &#x201c;hu:m&#x201d; Line 4 in excerpt <monospace>220928_56</monospace> and the voice lengthening on the determiner &#x201c;the&#x201d; Line 6 in excerpt <monospace>220324_79</monospace>. The fact that the robot does not align on the use of such turn-holding devices is the source of the failure, as the user claims the right to use these. Furthermore, within-turn silent pauses are made relevant by the use of such devices. Ideally, if the robot was more responsive and could handle overlap, a more subtle repair initiation (such as &#x201c;huh?&#x201d;) could be used in place, anticipating the possibility of an overlap. As said in <xref ref-type="sec" rid="s4-2">Section 4.2</xref>, overlap among humans is not a phenomenon to strictly avoid but to manage in real time, and so is the negotiation of turn-ending (<xref ref-type="bibr" rid="B104">Schegloff, 2002</xref>). Therefore, such online small feedback from the robot may be interpreted by the users as relevant in both cases (complete turn or incomplete turn). Online non-verbal feedback has been tested successfully in <italic>HRI</italic> in order to help generating the expected input (<xref ref-type="bibr" rid="B91">Pitsch et al., 2013</xref>).</p>
<p>The situation in which a robot is placed in a public place where users do not know its purpose and its functioning make these turn-holding devices ubiquitous. The cues made available by the human in order to overcome such sequential failure are the following:<list list-type="simple">
<list-item>
<p>&#x2022; The early display of alignment by the user: the (pre-) beginning of the turn shows that the user will provide the type-conforming answer;</p>
</list-item>
<list-item>
<p>&#x2022; The recognition of turn-holding devices that are used;</p>
</list-item>
<list-item>
<p>&#x2022; Syntactic completion.</p>
</list-item>
</list>However, syntactic completion would not be a useful cue in the case below. The human, this time, is re-enacting its turn as an overlap repair:</p>
<p>
<monospace>220928_19</monospace>
</p>
<p>
<monospace>01&#x2003;human1:&#x2003;bonjour peppe:r</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hello peppe:r</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.9)</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;((laughter))</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;((human1 leans towards pepper))</monospace>
</p>
<p>
<monospace>03&#x2003;human1:&#x2003;[o&#xf9;</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[where</monospace>
</p>
<p>
<monospace>04&#x2003;robot:&#x2003;&#x2003;[tu cherches quelque chose/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[are you looking for something/</monospace>
</p>
<p>
<monospace>05&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.9)</monospace>
</p>
<p>
<monospace>06&#x2003;human1&#x2003;&#x2003;[euh: ]</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[uh: ]</monospace>
</p>
<p>
<monospace>07&#x2003;human2:&#x2003;[l- l&#x2019;&#xe9;t]age (0.4) l&#x2019;&#xe9;tageavec le s[port/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[the flo]or (0.4) the floorwith s[port/</monospace>
</p>
<p>
<monospace>08&#x2003;robot:&#x2003;[pour te rendre aux &#xe9;tagessup&#xe9;rieurs</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;[in order to access upstairs</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;((robot continues))</monospace>
</p>
<p>Here, the syntactic completion criterion is validated with &#x201c;l&#x2019;&#xe9;tage&#x201d; (which can be translated as either &#x201c;the floor&#x201d; or &#x201c;upstairs&#x201d;). The state machine can provide an answer to the question &#x201c;how to get upstairs&#x201d;, and that is the response that is triggered here. But the program should also be designed to provide the floor number associated with subject-based sections (such as the floor number where the sports books are located). It is this second type of request that the human asks the robot (Line 7). Here, the robot failed in that it processed the chunk of audio delimited by the 0.4 silence (Line 7) and classified it (as a &#x201c;how-to-get-upstairs&#x201d; question) while the turn was actually continuing (as a &#x201c;subject X floor&#x201d; question).</p>
<p>Here, what leads to the formatting of the request on Line 7 is that two humans are competing for the floor in order to produce the request. After the robot&#x2019;s offer (Line 4), human1 claims the right for the turn with the &#x201c;uh&#x201d; (Line 6) as he leans towards Pepper (from Line 2) and already tried to formulate a request (Line 3) after the greeting (Line 1). Human2 also claims the right for the turn in overlap (Line 7). The silence of 0.4 s in the middle of his turn is the repair of his overlap with human1. Thus, such silences can be classified as an overlap repair if the information that &#x201c;two humans with different voices are talking at the same time&#x201d; is detectable.</p>
<p>Therefore, the cues made available by the human in order to overcome this type of sequential failure are the following:<list list-type="simple">
<list-item>
<p>&#x2022; The initial (multimodal) display of alignment by the users (as above);</p>
</list-item>
<list-item>
<p>&#x2022; The recognition of overlap (among humans) as the context for the turn processing (the function of the silence).</p>
</list-item>
</list>
</p>
</sec>
<sec id="s4-4">
<title>4.4 Two possible next actions</title>
<p>In this last case, the sequential failure is very simple to identify: the robot did not provide an answer (Line 7) to what might appear to be a simple move by the speaker (an offer acceptance). The robot was programmed to recognize the &#x201c;yes&#x201d; word in this state, however this word (Line 6) is part of a multi-unit turn (not separated by gaps) that plays a role in the sequential organization of such an opening. This raises a practical problem with regard to sequentiality: once identified, the initiating actions cannot be treated as a batch process of queued tasks. In 190 interactions that have been systematically annotated with regards to adjacency pairs, 163 pair parts (out of 1,417) are produced while there already is another expectancy for sequence completion. This is the case in the transcript below.</p>
<p>
<monospace>220321_06</monospace>
</p>
<p>
<monospace>01&#x2003;robot:&#x2003;&#x2003;coucou\ (.) je peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;hi\ (.) can I help you/</monospace>
</p>
<p>
<monospace>02&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(0.3)</monospace>
</p>
<p>
<monospace>03&#x2003;human2:&#x2003;.tsk .h coucou\</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;.tsk .h hi\</monospace>
</p>
<p>
<monospace>04&#x2003;human2:&#x2003;[((laugh))]</monospace>
</p>
<p>
<monospace>05&#x2003;human1:&#x2003;[((laugh))]</monospace>
</p>
<p>
<monospace>06&#x2003;human2:&#x2003;tu vas bien/ oui tu peuxm&#x2019;aider:\</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;how are you/ yes you can helpme\</monospace>
</p>
<p>
<monospace>07&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(1.5)</monospace>
</p>
<p>
<monospace>08&#x2003;human2:&#x2003;si tu m&#x2018; r&#xe9;ponds pas</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;if you don&#x2019;t answer me</monospace>
</p>
<p>
<monospace>09&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;(2.0)</monospace>
</p>
<p>
<monospace>10&#x2003;robot:&#x2003;&#x2003;comment est-ce que je peux t&#x2019;aider/</monospace>
</p>
<p>
<monospace>&#x2003;&#x2003;&#x2003;eng&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;how can I help you/</monospace>
</p>
<p>In Line 1, Pepper produces two <italic>First Pair Parts</italic> in the same turn: a greeting (&#x201c;hi&#x201d;) and a generic offer (&#x201c;can I help you?&#x201c;). This packing of two First Pair Parts is relevant and common in desk service encounter openings. These actions project two conditional relevancies: a return greeting, but also an offer acceptance OR rejection, OR some request/question (<xref ref-type="bibr" rid="B54">Kendrick and Drew, 2014a</xref>, p.101). In Line 3, the human produces the projected second greeting. That is, at the end of the turn Line 3, there is only one projected action (a response to the offer) that is still relevant. After the laughter (Lines 4&#x2013;5), the human produces (Line 6) a non-projected new action, a <italic>First Pair Part</italic> in the first segment of her turn (&#x201c;how are you?&#x201c;) which creates a new projection (a reciprocal response). Within the same turn, the projected offer acceptance is finally produced (&#x201c;yes you can help me&#x201d;). At this point (end of turn Line 6), the Second Pair Part made relevant by the offer is accomplished, but relevantly for a service encounter, another sequence is now projected either from the robot initiation (a proposal or second offer), or the human may now produce a request.</p>
<p>This means, that in Line 7 during the silence, two courses of action are again ongoing, with two types of projections on the next turns. On the one hand the &#x201c;how are you&#x201d; in l.6 projects a second pair part that is a reciprocal <italic>howareyou</italic> (<xref ref-type="bibr" rid="B105">Schegloff, 2007</xref>, pp.195&#x2013;202). On the other hand, the overall structural organization makes relevant a second offer/proposal from the robot or a request from the human. The return &#x201c;how are you&#x201d; is more relevant than the second offer as its relevancy is settled by a first pair part (<xref ref-type="bibr" rid="B112">Stivers and Robinson, 2006</xref>).</p>
<p>The utterance on Line 8 designates the preceding silence as a failure. Thus Pepper has the right to treat it as the initiation of a repair sequence (in other words, a correction of the previous turn/action), which again makes conditionally relevant a repair completion. Also, at this point in the unfolding of the interaction, ostensibly the human has withdrawn the possibility to produce the request herself (as hypothetically expected from service encounter norms), as the turn-allocation to Pepper is reinforced. On Line 9, another long silence shows that the human maintains the possibility for Pepper to produce any next action that could align with the possibilities that we pointed to. On Line 10, finally, Pepper produces the possible (among other actions) second offer, designed as a question that more clearly pursues the production of a request from the human. At this point, the repair is completed.</p>
<p>As we can see, the sequential organisation of interaction can imply different possible next actions. The embodiment of such actions in time has a great impact on the interaction. Here, what can be leveraged from sequential organization in order to process such talk-in-interaction is the fact that:<list list-type="simple">
<list-item>
<p>&#x2022; There are two actions in the same turn (without even a small pause);</p>
</list-item>
<list-item>
<p>&#x2022; A &#x201c;how are you&#x201d; is relevant after the greeting exchange;</p>
</list-item>
<list-item>
<p>&#x2022; The fact that possible next actions, once identified, can be ordered: the expectancy is not the same turn-to-turn, sequence-to-sequence, with regards to overall structural organization.</p>
</list-item>
</list>
</p>
<p>The fact that the interaction order is operating at multiple levels of granularity is also the source of error in the following excerpt:</p>
<p>
<monospace>220317_24</monospace>
</p>
<p>
<monospace>01&#x2003;human:&#x2003;where can I find books about math?</monospace>
</p>
<p>
<monospace>02&#x2003;robot:&#x2003;((provides information as aresponse to the user&#x2019;s question))</monospace>
</p>
<p>
<monospace>03&#x2003;robot:&#x2003;is that clear?</monospace>
</p>
<p>
<monospace>04&#x2003;human:&#x2003;yes thanks</monospace>
</p>
<p>
<monospace>05&#x2003;robot:&#x2003;okay, I will repeat ((pepperrepeats turn L.2))</monospace>
</p>
<p>Here, the robot recognized &#x201c;no thanks&#x201d; (l.4)<xref ref-type="fn" rid="fn8">
<sup>8</sup>
</xref>: it thus repeats the answer to the user&#x2019;s question (l.5). Differentiating &#x201c;no thanks&#x201d; vs. &#x201c;yes thanks&#x201d; is difficult for an ASR in a noisy environment. But these two possibilities are not equal with regards to what it accomplishes in the interaction. The &#x201c;no thanks&#x201d; accomplishes one action (a rejection registered as such). Here again, what is relevant is the ordering of possibilities: the &#x201c;thanks&#x201d; retrospectively indexes and expands the previous question/answer sequence (ll.1&#x2013;2). The human first answered &#x201c;yes&#x201d; to the confirmation request (l.3) but also thanks the robot for its service (provided l.2) at the first possible slot.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Towards natural human-robot interaction</title>
<p>To overcome (some of) the limitations observed in the previous sections (such as speaking out of turn (<xref ref-type="sec" rid="s4-3">Section 4.3</xref>) and the shallow/insufficient treatment of both two-action turns (<xref ref-type="sec" rid="s4-1">Sections 4.1</xref> and <xref ref-type="sec" rid="s4-2">4.2</xref>) and multiple active adjacency pairs (<xref ref-type="sec" rid="s4-4">Section 4.4</xref>)), we require a model that can 1) make more sophisticated use of turn-taking and turn-holding cues, 2) identify more than one action within a turn, 3) learn to handle multiple threads in the interaction and 4) incorporate multi-modal features into its context understanding. In this section, we will delve into each of these requirements and review strategies that have been proposed to address these complexities as best as possible.</p>
<p>The limitations so described and the way to address them is very much affected by the ability of a dialogue manager to capture the flow and meaning of conversations. Indeed, if a model can comprehend these factors, it can easily 1) infer the contextual end of turns, 2) &#x26; 3) identify actions and threads that are not associated in a random way but aligned with the meaning of the conversation.</p>
<p>In dialogue systems, the current state of the discourse can be modelled using rule-based, frame-based or end-to-end systems. Rule-based and frame-based approaches are more directed toward the expertise of the human designer, while end-to-end systems are based on statistics extracted from a large corpus of texts and can be refined or oriented thanks to a pre-prompting strategy in order to fit the application needs. In this section, we first introduce and discuss these two popular families of methods to catch the contextual meaning and discuss inherent drawbacks. We then cover the approaches from the literature that can mitigate the limits described to a certain extend. Many of the approaches are hybrid and borrow both from hand-made settings and automatic learning.</p>
<p>There are of course other limitations that prevent smooth interaction between Pepper and the users, for example computational latency and the noisy environment which degrades automatic speech recognition. In this work, however, we choose to focus on the elements pertaining to turn-taking and conditional relevancy.</p>
<sec id="s5-1">
<title>5.1 Contextualisation as a support for human-robot interaction</title>
<sec id="s5-1-1">
<title>5.1.1 Handcrafted approaches</title>
<p>As a task-oriented conversational agent, the robot is set up to provide answers for each targeted question. One way to address functionality is rooted in reproducing what is expected to be seen, to mimic what a human would do. The general principle being that the model contains expert knowledge which guides the behaviour of the system. This can be done thanks to a set of human-made rules designed to make decisions and provide solutions for specific problems. It can also include other specific supports, such as graphs, to map interactions.</p>
<p>This family of approaches can offer numerous benefits, such as intelligibility and easy updating. They nevertheless suffer from drawbacks including their inability to handle unexpected situations that have not been modelled. It is very difficult to model all the sequences that can occur by hand, especially when it involves semantics and conversations. In our experiments, interactions can easily exceed the scope the system is prepared for. Many of the approaches covered below follow the same logic, in particular those used in multi-thread modelling (<xref ref-type="bibr" rid="B99">Ros&#xe9; et al., 1995</xref>; <xref ref-type="bibr" rid="B64">Lemon et al., 2002</xref>; <xref ref-type="bibr" rid="B57">Kl&#xfc;wer, 2015</xref>; <xref ref-type="bibr" rid="B77">Maraev et al., 2020</xref>). Finite-state machines (FSM), like what is used by Pepper, belong to this category.</p>
<p>Complications ensue in FSMs because the interaction does not always adhere to a strict set of succinct linear moves. We see this when the user performs two actions within the same turn; the broader the domain, the more difficult it becomes to cover with precise rules every possible combination of actions. Furthermore, smooth transitions to future states can only be achieved if the user provides an expected response; any deviation from the conversation design or complexity in the formulation of a turn will likely result in failure (managing multiple threads was however made possible to some extent in this paradigm by using a hierarchical structure with sub-automata (<xref ref-type="bibr" rid="B57">Kl&#xfc;wer, 2015</xref>, further details below).</p>
</sec>
<sec id="s5-1-2">
<title>5.1.2 Statistical approaches</title>
<p>As for statistical approaches, methods that incorporate pretrained language models have the greatest potential to expand a model&#x2019;s contextual understanding since they are able to represent complex states within a latent space. In recent years, large language models (LLMs) have made significant progress in natural language processing. A survey of capacities for such systems and how they work is provided by <xref ref-type="bibr" rid="B120">Wang et al. (2024)</xref>. These systems can either be used in an end-to-end fashion or they can be applied to the modules of the standard dialogue system pipeline (i.e., natural language understanding, dialogue management, natural language generation). An end-to-end system can be prompted with conversation history, and can then generate possible developments in the conversation, i.e. the user responses. They are trained on a general purpose large corpus, and can be later fine-tuned for specific applications. A pre-prompt can be included in the input in order to specify the expected responses (style, length, allowed domains, etc.).</p>
<p>These systems aim to implicitly handle, in the way they work, two of the three problems identified: multi-intent detection and multi-thread management. Indeed, for these approaches, there are no assumptions about the content of the conversation in such a way that the system will generate a probable answer in relation to what it has already seen during its training. These approaches can take as input the history of the interaction, which is really suited to keeping track of the dialogue state (e.g. <xref ref-type="bibr" rid="B122">Wen et al., 2017</xref>; <xref ref-type="bibr" rid="B38">Ham et al., 2020</xref>; <xref ref-type="bibr" rid="B45">Imrattanatrai and Fukuda, 2023</xref>).</p>
<p>Although, generative AI is very impressive, it nevertheless suffers from inherent limitations. LLM&#x2019;s ability to generalize, while a strength when it comes to understanding previously unseen contexts, can also be a double-edged sword, as generalizations can lead to hallucinations (e.g. <xref ref-type="bibr" rid="B126">Yamazaki et al., 2023</xref>). If the model encounters a question for which it has no answer, it may simply invent a response which shares characteristics of content it has seen in its training data, but which has no factual validity. There may also be issues with response consistency (i.e., asking the model the same question twice could result in different and/or contradictory responses, see <xref ref-type="bibr" rid="B110">Song et al., 2020</xref>; <xref ref-type="bibr" rid="B51">Kassner et al., 2021</xref>). For a task oriented system, it is very important to have precise control over the information delivered to the user and so steps must be taken to rein in its generation. Compensation mechanisms proposed in the literature include knowledge grounding (<xref ref-type="bibr" rid="B70">Lin et al., 2022</xref>; <xref ref-type="bibr" rid="B114">Sun et al., 2023</xref>) and fine-tuning (<xref ref-type="bibr" rid="B86">Nguyen et al., 2023</xref>). LLM&#x2019;s are nevertheless difficult to restrain when the corpus used for training does not correspond exactly to the situation to manage. For example, an LLM could take the initiative and offer to accompany the user into the library, even though the robot is not equipped to move around independently. In a way, when describing what to do for every use case, thanks for instance to fine tuning or pre-prompting strategies, we come up against the same drawbacks as the ones of descriptive strategies.</p>
<p>A further drawback of LLMs is that the majority of them are trained on textual data alone, and when applied to managing embodied interaction, they will not take into consideration important multimodal features (e.g., intonation, gesture, laughter, etc.) which modify expectations about appropriate future turns (although there is increasing interest in incorporating such features, see e.g. <xref ref-type="bibr" rid="B20">Driess et al., 2023</xref>; <xref ref-type="bibr" rid="B44">Huang et al., 2023</xref>; <xref ref-type="bibr" rid="B55">Kharitonov et al., 2022</xref>, so these issues may be overcome in the near future).</p>
</sec>
</sec>
<sec id="s5-2">
<title>5.2 Turn taking dynamics</title>
<p>A number of the observed failures could be better handled with improved turn taking skills. Pepper&#x2019;s system relies purely on silence to detect the end of the user&#x2019;s turn, which is clearly insufficient because it does not make the distinction between a within-turn pause (<xref ref-type="bibr" rid="B39">Harvey Sacks and Jefferson, 1974</xref>) and turn yielding. More sophisticated approaches incorporate a wider array of features to detect whether the user&#x2019;s turn has come to completion. These signals include verbal, prosodic, breathing, gaze and gesture cues.</p>
<p>Verbal cues, including syntactic, semantic and pragmatic features, are important cues for human turn-end prediction (<xref ref-type="bibr" rid="B29">Ford and Thompson, 1996</xref>; <xref ref-type="bibr" rid="B30">Ford et al., 1996</xref>). Simple models that use only the part-of-speech of the final two words (<xref ref-type="bibr" rid="B36">Gravano and Hirschberg, 2011</xref>; <xref ref-type="bibr" rid="B79">Meena et al., 2014</xref>) are often able to detect incomplete turns as certain categories (e.g., determiners without the following noun) are unlikely to be the end of a contribution.<xref ref-type="fn" rid="fn9">
<sup>9</sup>
</xref> In fact, leaving an utterance syntactically incomplete is a strategy employed by speakers in order to hold the floor (<xref ref-type="bibr" rid="B107">Selting, 2000</xref>).</p>
<p>As we saw in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>, syntactic completeness/incompleteness alone would be insufficient to capture all the failures present in our corpus. Large language models could provide richer syntactic, semantic and pragmatic information for end-of-turn prediction. <xref ref-type="bibr" rid="B24">Ekstedt and Skantze (2020)</xref> adapt a GPT-2 model to include transition relevance place (TRP) tokens as part of the model&#x2019;s vocabulary and this method provides significant gains over simpler POS-based ones. Further progress could be achieved by incorporating knowledge of sequential context (see <xref ref-type="sec" rid="s4-3">Section 4.3</xref>).</p>
<p>It has been an issue of debate whether prosodic cues are a necessary component of turn-end identification. <xref ref-type="bibr" rid="B100">Ruiter et al. (2006)</xref> removed pitch contours from recorded English conversations and this appeared to have little effect on human prediction accuracy. <xref ref-type="bibr" rid="B8">B&#xf6;gels and Torreira (2015)</xref>, however, found that prosody was crucial for distinguishing between holding an yielding at ambiguous TRPs. <xref ref-type="bibr" rid="B26">Ekstedt and Skantze (2022)</xref> found similar results.</p>
<p>Non-linguistic features can also improve prediction accuracy. Exhaling has been associated with turn yielding and inhaling with turn retention (<xref ref-type="bibr" rid="B98">Rochet-Capellan and Fuchs, 2014</xref>; <xref ref-type="bibr" rid="B46">Ishii et al., 2014</xref>). Gaze helps regulate floor management (<xref ref-type="bibr" rid="B15">Degutyte and Astell, 2021</xref>) as, for example, looking away and, then, doing a return gaze, are conducts involved in building multi-unit turns, holding the floor, and accomplishing floor transfer (<xref ref-type="bibr" rid="B52">Kendon, 1967</xref>; <xref ref-type="bibr" rid="B35">Goodwin, 1981</xref>). This can however be complicated by other factors such as focus on reference items or the presence of multiple participants (<xref ref-type="bibr" rid="B50">Jokinen et al., 2013</xref>; <xref ref-type="bibr" rid="B7">Bilac et al., 2017</xref>). Gesture also plays a role, as turn completion will often coincide with movement completion (<xref ref-type="bibr" rid="B21">Duncan, 1972</xref>).</p>
<p>An ideal system would be able to anticipate the end of their interlocutor&#x2019;s turn, as humans do. Evidence for this comes from the fact that human&#x2019;s typically begin their turn within 200 ms of the end of the previous turn (<xref ref-type="bibr" rid="B66">Levinson and Torreira, 2015</xref>). This is insufficient time to plan and execute a new turn, suggesting that the listener both projected the end of the speaker&#x2019;s turn and planned their turn in advance. They are able to do this projection because the previous turn has created expectations about what the current turn should entail. To replicate this behaviour for human-robot interaction, <xref ref-type="bibr" rid="B25">Ekstedt and Skantze (2021)</xref> use Turn-GPT (<xref ref-type="bibr" rid="B24">Ekstedt and Skantze, 2020</xref>) to project the end of the user&#x2019;s turn by generating several possible continuations for the speaker&#x2019;s utterance. If a sufficient number of these continuations project an ending, the model would prepare to take the floor. This method offered an improvement over a silence baseline in terms of reduction of both speech overlap and extended silences.</p>
<p>In our corpus analysis, we also observe failures in grounding (i.e., where the fact that Pepper is processing an action of the user is not properly communicated to the user). When the user interprets Pepper&#x2019;s silence as a failure of hearing or understanding, they are likely to start a new turn, which can result in overlapping speech and a misalignment of user and agent expectations about upcoming speech. Adding fillers to robot speech has been shown to have a positive effect on the perceived speediness of the agent (<xref ref-type="bibr" rid="B123">Wigdor et al., 2016</xref>).</p>
</sec>
<sec id="s5-3">
<title>5.3 Multi-intent detection</title>
<p>As we have seen in <xref ref-type="sec" rid="s4-2">Sections 4.2</xref>, one turn by the user does not always correspond with one intention, which can be a major source of errors for a system designed/trained to only look for one intention per turn. Simple solutions, such as taking the top-k classes predicted by a single intent classifier do not yield high results (<xref ref-type="bibr" rid="B125">Xu and Sarikaya, 2013</xref>) and so specialized multi-label models have been developed (<xref ref-type="bibr" rid="B94">Qin et al., 2020</xref>; <xref ref-type="bibr" rid="B84">Natarajan et al., 2020</xref>; <xref ref-type="bibr" rid="B12">Cheng et al., 2023</xref>).</p>
<p>
<xref ref-type="bibr" rid="B56">Kim et al. (2017)</xref> looked for overt lexical markers (e.g. conjunctions) to identify possible divisions within the user&#x2019;s utterance; once the sentence was split, single intent classifiers could be applied to the sentence parts. This method is quite limited though, as multiple intents sentences are not always so neatly demarcated and sometimes treating them separately can have negative consequences (e.g., <italic>Find Avatar/and/play it.</italic>).</p>
<p>Other research has investigated the joint prediction task of multiple intent with slot filling (<xref ref-type="bibr" rid="B31">Gangadharaiah and Narayanaswamy, 2019</xref>; <xref ref-type="bibr" rid="B94">Qin et al., 2020</xref>; <xref ref-type="bibr" rid="B84">Natarajan et al., 2020</xref>; <xref ref-type="bibr" rid="B111">Song et al., 2022</xref>). This design acknowledges the possibility that a once mentioned entity in the utterance could relate to different intents. <xref ref-type="bibr" rid="B111">Song et al. (2022)</xref> make use of global corpus statistics to learn explicit dependencies between intents and slots. Because the task of the classification is more difficult in a multi-label setting, error propagation can be a concern. To mitigate this, <xref ref-type="bibr" rid="B12">Cheng et al. (2023)</xref> propose a scope sensitive model which filters out words that are not semantically related to the intent classes.</p>
<p>Attempts have been made to leverage semantic similarities between intent classes to improve classification accuracy (<xref ref-type="bibr" rid="B125">Xu and Sarikaya, 2013</xref>; <xref ref-type="bibr" rid="B124">Wu et al., 2021</xref>). So rather than finding a mapping between user utterances and indexed intent categories, these methods find overlaps in meaning between categories such as getWeather and getTime (e.g., both are asking to retrieve information). <xref ref-type="bibr" rid="B125">Xu and Sarikaya (2013)</xref> model class features by predicting combined intent labels. <xref ref-type="bibr" rid="B124">Wu et al. (2021)</xref> learnt an intent semantic space by extracting the semantic information present in the intent labels. They then project the utterance embedding into the intent space and use linear approximation to learn the linear combination of the intent basis. This method can be extended to unseen intents during training, although the fine-grained distinctions between unseen classes is imperfect.</p>
<p>Dealing with multiple intents, once identified, can raise other issues for the system if two separate first pair parts have been put forward: a decision must be made about which action to deal with first. <xref ref-type="bibr" rid="B61">Landesberger and Ehrlich (2019)</xref> propose a six part strategy to prioritize responses to multi-intent turns: 1) explicit sequence ordering (e.g., <italic>First tell me when the party is, then phone my mother.</italic>), 2) thematic dependency (i.e., where the accomplishment of one task is dependant on the prior accomplishment of the other), 3) urgency, 4) efficiency and 5) personal preference (<xref ref-type="bibr" rid="B62">Landesberger and Ehrlich, 2020</xref>, show that prosodic features are able to detect the level of urgency within multi-intent turns).</p>
<p>
<xref ref-type="bibr" rid="B60">Landesberger and Ehrlich (2018)</xref> also investigated user strategies for managing situations where there has been a misunderstanding in one of the intents from a multi-intent turn. Users displayed different behaviours, either addressing the system&#x2019;s question and then providing a correction or only addressing the correction. A well designed model must be able to handle both of these possible reactions.</p>
</sec>
<sec id="s5-4">
<title>5.4 Multi-thread management</title>
<p>In the most basic scenario for task-oriented dialogue, either the user or the system will initiate a <italic>First Pair Part</italic> and this will immediately be followed by a <italic>Second Pair Part</italic> (e.g., question/response). However, in practice other acts may intervene before the <italic>Second Pair Part</italic> is completed, as we saw in <xref ref-type="sec" rid="s4-4">Sections 4.4</xref> and <xref ref-type="sec" rid="s4-1">4.1</xref>. This presents challenges for dialogue management: the system must be able to 1) represent active threads and 2) make decisions about which threads to pursue and in which order, as well as which to abandon.</p>
<p>Multi-threaded dialogue can refer to either embedded sequences (e.g., clarification questions) or interleaved ones (i.e., utterances pertaining to different tasks that are intermingled). Proposals to handle multiple active strands have involved extendable graph representations of the ongoing discourse and/or task stacks, where the most recently active thread is prioritized but incoming utterances can still be linked to lower level threads present on the stack, or they can create their own branch (<xref ref-type="bibr" rid="B99">Ros&#xe9; et al., 1995</xref>; <xref ref-type="bibr" rid="B64">Lemon et al., 2002</xref>; <xref ref-type="bibr" rid="B57">Kl&#xfc;wer, 2015</xref>; <xref ref-type="bibr" rid="B87">Papaioannou et al., 2018</xref>; <xref ref-type="bibr" rid="B77">Maraev et al., 2020</xref>). For example, <xref ref-type="bibr" rid="B64">Lemon et al. (2002)</xref> use a dialogue move tree to represent the dialogue state and an active node list to represent the order of the most recently activated threads. If the incoming input satisfies an update function for one of the nodes on the stack, it is attached to that node. Similarly, <xref ref-type="bibr" rid="B57">Kl&#xfc;wer (2015)</xref> propose a dialogue manager that represents conversation threads (implemented as supernodes) as having one of three conditions: active, paused and inactive. If a suitable transition from the currently active thread is found lacking, then either a paused or a new thread is activated and this selection is done by considering the topic, dialogue act and domain.</p>
<p>
<xref ref-type="bibr" rid="B108">Shi et al. (2019)</xref> observe that certain user queries are often followed by related queries, the results of which can cause the user to return to the initial task (for example, asking to schedule a meeting on a given date, followed by a weather check for that same date, and then revising/updating the date for the meeting to a new date with more suitable weather). In order to anticipate the users&#x2019; needs and reduce redundancies in query formulation and bolster intent classification, they propose a model that predicts whether the user is likely to switch topics and if so, they proactively provide the information.</p>
<p>Other works have incorporated the sequences identified in conversational analysis into their system design. In the Natural Conversational Framework (NCF) (<xref ref-type="bibr" rid="B81">Moore and Arar, 2019</xref>), interactions are designed as expandable sequences which can accommodate expansions such as clarifications and repairs. <xref ref-type="bibr" rid="B59">Kunneman and Hindriks (2022)</xref> take the patterns outlined in NCF and develop a dialogue engine which keeps track of the status of sequences (complete/incomplete). <xref ref-type="bibr" rid="B22">Duran (2023)</xref> trained models to automatically annotate adjacency pair labels which can then be used by a dialogue manager to determine the next move.</p>
<p>Models for dialogue disentanglement have been developed to separate chat room, social media and forum threads where multiple participants are conversing in interwoven discussions (<xref ref-type="bibr" rid="B72">Liu H. et al., 2021</xref>; <xref ref-type="bibr" rid="B67">Li et al., 2023</xref>; <xref ref-type="bibr" rid="B37">Gu et al., 2021</xref>). This separation aids information extraction and summarization. State-of-the-art deep learning techniques have applied global discourse structure to accomplish the task, which would not be available when processing the discourse in a linear fashion, however earlier techniques (<xref ref-type="bibr" rid="B58">Kummerfeld et al., 2019</xref>; <xref ref-type="bibr" rid="B130">Zhu et al., 2021</xref>) that model reply-to relations between utterances using features such as time between utterances, word overlap and anaphoric links could be transferable to actional/sequential thread classification in dialogue systems.</p>
<p>Not all actions that have been started must necessarily be handled by the system as the changing interactional context may make them irrelevant. We saw this in <xref ref-type="sec" rid="s4-1">Section 4.1</xref> when a greeting (<italic>Hello Pepper</italic>) is immediately followed by closing sequence (<italic>Have a good day</italic>), making the expectation for a return greeting less pertinent. <xref ref-type="bibr" rid="B47">Janarthanam and Lemon (2014)</xref> implement a policy for discarding threads that are no longer relevant (Queue revision), although their model only considers the user&#x2019;s geographical location in a city as criteria for thread elimination and not interactional phenomena.</p>
<p>End-to-end systems for task-oriented dialogue (<xref ref-type="bibr" rid="B95">Qun et al., 2020</xref>; <xref ref-type="bibr" rid="B122">Wen et al., 2017</xref>)) allow the model to learn the types of sequences that appear naturally in the training corpus (which could include both interleaved and embedded sequences) without the model designer having to specify them. While there is a good chance large language models would be able to handle multiple threads, to the best of our knowledge, this has not been tested empirically and this would likely be dependent on the size/memory of the model, as well as the number of mixed sequences available in the training data.</p>
<p>Finally, it is important that the dialogue manager signals to the user which thread is being attended to. If the system is responding to anything but the most recently activated one, this could be confusing for the user. When humans switch between topics, they make use of discourse markers (<xref ref-type="bibr" rid="B40">Heeman et al., 2005</xref>; <xref ref-type="bibr" rid="B128">Yang et al., 2008</xref>). And when they return to a pre-existing thread, humans frequently restore the previous context with repetition (<xref ref-type="bibr" rid="B127">Yang and Heeman, 2009</xref>).</p>
</sec>
<sec id="s5-5">
<title>5.5 Multimodal cues</title>
<p>In order to facilitate an understanding of context, cues beyond the verbal need to be taken into consideration. Visual cues are a rich source of information for interaction, however from a technical point of view there are still challenges to overcome in dissecting intricate, evolving scenes in order to make them interpretable to the robot system.</p>
<p>A first step is to identify the presence of humans in the environment which can be accomplished using human detection neural models (<xref ref-type="bibr" rid="B78">Marvasti-Zadeh et al., 2021</xref>). More demanding is maintaining a consistent representation of the identified people, which is necessary to manage the context history (i.e., how long have I been talking to this person and what have we talked about). Not having mastered this skill, Pepper would often reintroduce himself mid-conversation during our experiment.</p>
<p>In a public space, where people are constantly moving in and out of the scene, people monitoring is not an easy task. Tracking algorithms have been developed (e.g. <xref ref-type="bibr" rid="B129">Zhang et al., 2022</xref>; <xref ref-type="bibr" rid="B13">Cheng et al., 2024</xref>), but these can break down when a person momentarily leaves or is occluded from a camera&#x2019;s view (<xref ref-type="bibr" rid="B73">Liu S. et al., 2021</xref>). If people can be tracked at a fairly reliable level however, then cues regarding their movements and proximity to the robot can be used to evaluate the type of engagement the human wants to engage in (e.g., a quick hello/goodbye vs. a prolonged conversation).</p>
<p>Active speaker detection models (<xref ref-type="bibr" rid="B80">Min et al., 2022</xref>; <xref ref-type="bibr" rid="B68">Liao et al., 2023</xref>; <xref ref-type="bibr" rid="B1">Alc&#xe1;zar et al., 2020</xref>) which use both audio and visual signals as well as speaker relations, could aid in the detection of overlapping speech. Dedicated models for such events have also been proposed using audio alone for diarization tasks (e.g. <xref ref-type="bibr" rid="B9">Bullock et al., 2020</xref>). Once detected, the robot system could allow for more time for the user&#x2019;s turn with the knowledge that a silence may not indicate the end of a turn, but a repair.</p>
<p>The visual is also important for embodied interaction as gesture (e.g., a head nod or a thumbs up) can be interpreted as a conversational turn in its own right: neural network architectures have made great strides in recent years in recognizing these actions (<xref ref-type="bibr" rid="B49">Ji et al., 2020</xref>).</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>In this work, we have identified failure types within a video-recorded corpus of human-robot interaction and, using the tools of conversational analysis, we have offered explanations for such communication difficulties. Experiments were carried out using a task-oriented conversational agent implemented on the robot Pepper to welcome and guide visitors in a library. When designing Pepper&#x2019;s scenario, we did not strictly follow the robot&#x2019;s guidelines, i.e., we did not use the display screen as recommended.<xref ref-type="fn" rid="fn10">
<sup>10</sup>
</xref> Rather, in this scenario, the user is expected to mostly rely on adapting human adjacency pairs and the interaction overall organization in the case of service encounters. In this manner, we compiled a corpus of in-the-wild human-robot interactions and identified concrete and typical cases of human behaviour and adaptations when confronted with such a system used in public.</p>
<p>This setup allows us to identify and deeply analyse some failures during interactions, taking the system&#x2019;s inability to contextualize turns as a central point. Our study is particularly interested in the way the context of failures can be described through the sequential organization of interaction, and methods from computer science to enhance the dialogue management in this regard.</p>
<p>The system used in our experiment cannot for instance use syntax and pragmatic reasoning to determine when the speaker has finished the current turn based on constraints imposed by the previous turn. Nor can it keep track of the opening and closing of adjacency pairs, when for example two actions are performed within the same turn or when a user switches threads after an issue with grounding.</p>
<p>Our analysis is based on an interpretation of the conversation in all its dynamism and sequentiality, rather than on a more local one. When analysing failures in the analytic section, we accordingly showed that the cues that the robot needs to identify can only be understood with the knowledge that interaction is managed through a sequential organisation of expectancy. This underlines the central role of contextualisation in dialogue modelling.</p>
<p>Because these cues are purposely made visible by participants (drawing from human-human interaction), we propose that they can be objectified and computed. This raises the question of how to compute these in accordance with the sequential organization that we highlighted. We present this (based on CA theory) as a normative organization and not a rule set, which consequently cannot be easily emulated with a rule-based approach. As highlighted in the theoretical section (<xref ref-type="sec" rid="s2">Section 2</xref>), some of the human-human interaction norms hypothetically elicited by the robot may be rejected or welcomed by the users as reified rules for participating with it.</p>
<p>Through our analysis, where we bring to light the sequential and actional bonds between (parts of) turns, we identify events as instances of classes we define incrementally, taking place over the course of time. Whether or not it is possible to have better results through a statistical approach with regards to sequencing remains to be tested.</p>
<p>Failures in our experiment stem from the system&#x2019;s inability to contextualize turns. The ability of models to build a relevant and appropriate representation or contextualization from a signal (such as speech/text) has been a long-standing concern in AI. This issue has been addressed through descriptive or expert approaches, and more recently, through machine learning, where large language models have proven effective. After describing the advantages and drawbacks of each of these paradigms, we review the literature for each of the highlighted causes of failure, namely turn-taking, multi-intent identification, multi-thread handling and multi-modal understanding.</p>
<p>The most promising of the covered methods are those that offer flexibility and robustness when faced with a broad range of different contextual states and complex user inputs. The use of large language models in conversational agents has great potential to overcome the observed failures, if controls can be put in place to control their generation. In contrast, descriptive approaches are more deterministic and predictable but may struggle to adapt beyond the specific frameworks for which they were designed.</p>
<p>This study raises a number of questions and challenges for dialogue modelling. Some of the studied patterns have not yet been investigated in state-of-the-art models (to the best of our knowledge). This is the case for multi-thread management, for which an LLM end-to-end approach appears to have great potential. While these models are promising, for some specific problems, LLMs may struggle to capture the contextual meaning of underlying structures present in the corpus. This is likely to be particularly pronounced when the training corpora are not dedicated to human-robot interaction, which can introduce biases. The corpus we collected has been labeled for each of the failure types, such that it could be used to probe an LLM&#x2019;s latent representations (for example, by testing its ability to maintain context across multiple conversational threads or correctly identify multiple intents in a single interaction). These probes could then contribute to the building of a more advanced and relevant conversational agent.</p>
<p>The described work has been conceived independently from the application domain. As HRI research identifies failure types of different kinds, there is a need to build test scenarios that could be used to evaluate specific devices or use cases. We advocate that an embodied HRI scenario should elicit the situations we presented, as the interactional outcomes that we analyzed are ubiquitous and therefore should be handled adequately.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s8">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Sylvie Collignon D&#xe9;l&#xe9;gu&#xe9;e &#xe0; la Protection des Donn&#xe9;es (DPD) au CNRS. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study. Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec id="s9">
<title>Author contributions</title>
<p>LT: Conceptualization, Data curation, Formal Analysis, Methodology, Writing&#x2013;original draft, Writing&#x2013;review and editing, Investigation, Resources, BS: Conceptualization, Formal Analysis, Investigation, Methodology, Resources, Writing&#x2013;original draft, Writing&#x2013;review and editing, HB-Q: Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Validation, Writing&#x2013;original draft, Writing&#x2013;review and editing. ML: Conceptualization, Writing&#x2013;review and editing. FA: Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Validation, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The authors are grateful to the ASLAN project (ANR-10-LABX-0081) of the Universit&#xe9; de Lyon, for its financial support within the French program &#x201c;Investments for the Future&#x201d; operated by the National Research Agency (ANR).</p>
</sec>
<ack>
<p>The authors wish to thank Antoine Bouquin, the research engineer who programmed the state machine that allowed the recordings. In particular, our discussions with Antoine allowed us to understand and anticipate the limits of this system.</p>
</ack>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The handling editor FF declared a past co-authorship with the author LT.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://qisdk.softbankrobotics.com/sdk/doc/pepper-sdk/index.html">https://qisdk.softbankrobotics.com/sdk/doc/pepper-sdk/index.html</ext-link>
</p>
</fn>
<fn id="fn2">
<label>2</label>
<p>ELAN (Version 6.7) [Computer software] (2023). Nijmegen: Max Planck Institute for Psycholinguistics, The Language Archive. Retrieved from <ext-link ext-link-type="uri" xlink:href="https://archive.mpi.nl/tla/elan">https://archive.mpi.nl/tla/elan</ext-link>
</p>
</fn>
<fn id="fn3">
<label>3</label>
<p>
<ext-link ext-link-type="uri" xlink:href="http://icar.cnrs.fr/corinte/conventions-de-transcription/">http://icar.cnrs.fr/corinte/conventions-de-transcription/</ext-link>
</p>
</fn>
<fn id="fn4">
<label>4</label>
<p>Another issue here is the fact that the self-presentation performed by Pepper conflicts with the recognition displayed by the human. However, the recognition and accomplishment of &#x201c;greetings alone&#x201d; exchanges between acquainted people supersedes this issue (that is, treating the address term &#x201c;Pepper&#x201d; as a cue for such an activity distinction would be of very low importance).</p>
</fn>
<fn id="fn5">
<label>5</label>
<p>In fact, by the categorization of Pepper as &#x201c;a regular&#x201d; vs. a service provider, and as evidenced by the additional cases (and particularly <monospace>220328_06A</monospace>) where laughter follows the greeting produced in the move, it can be considered as a &#x201c;successful failure&#x201d; from the user perspective, as proposed in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>.</p>
</fn>
<fn id="fn6">
<label>6</label>
<p>If the system matches the fact that a place is &#x201c;looked for&#x201d; but the system does not recognize the place, it can then initiate a repair. Then, by responding to the repair initiated by the robot, users may isolate the name of the place they are looking for, thereby easing the recognition on the second attempt. The &#x201c;repair list&#x201d;, thus, has a lower priority on place recognition.</p>
</fn>
<fn id="fn7">
<label>7</label>
<p>Another interesting event is observed in excerpt <monospace>220324_79</monospace>, another event is not mentioned here it is the laughter that accomplishes a type of sequence-closing third assessment often seen in HRI (see also excerpt <monospace>220321_06</monospace> in <xref ref-type="sec" rid="s4-4">Section 4.4</xref> or <monospace>220928_19</monospace> in this section for the exact same events). Handling such interaction between humans is another issue that is relevant with regard to sequentiality but we will not address this here. It is also a good example of specific HRI practices (<xref ref-type="bibr" rid="B69">Licoppe and Rollet, 2020</xref>).</p>
</fn>
<fn id="fn8">
<label>8</label>
<p>Note that ideally, here, in this specific context, the wordlist containing &#x201c;no thanks&#x201d; should not be able to match at all, but this does not entail the rest of the comment as such this situation can be reproduced in other contexts, such as following an offer.</p>
</fn>
<fn id="fn9">
<label>9</label>
<p>An exception to this principle does occur when the speaker is eliciting the listener to finish their utterance when they cannot think of the word (<xref ref-type="bibr" rid="B14">Clark and Wilkes-Gibbs, 1986</xref>).</p>
</fn>
<fn id="fn10">
<label>10</label>
<p>
<ext-link ext-link-type="uri" xlink:href="http://doc.aldebaran.com/download/Pepper_B2BD_guidelines_Sept_V1.5.pdf">http://doc.aldebaran.com/download/Pepper_B2BD_guidelines_Sept_V1.5.pdf</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Alc&#xe1;zar</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Caba</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Perazzi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.-Y.</given-names>
</name>
<name>
<surname>Arbel&#xe1;ez</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Active speakers in context</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>12465</fpage>&#x2013;<lpage>12474</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Aoki</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Szymanski</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Plurkowski</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Thornton</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Woodruff</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>Where&#x2019;s the party in multi-party? analyzing the structure of small-group sociable talk</article-title>,&#x201d; in <source>Proceedings of the 2006 20th anniversary conference on Computer supported cooperative work</source>, <fpage>393</fpage>&#x2013;<lpage>402</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arend</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sunnen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Caire</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Investigating breakdowns in human robot interaction: a conversation analysis guided single case study of a human-nao communication in a museum environment</article-title>. <source>Int. J. Mech. Aerosp. Industrial, Mechatron. Manuf. Eng.</source> <volume>11</volume>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Avgustis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Shirokov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Iivari</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>&#x201c;Please connect me to a specialist&#x201d;: scrutinising &#x2018;recipient design&#x2019; in interaction with an artificial conversational agent</article-title>,&#x201d; in <source>Interact 2021</source> (<publisher-name>Springer Nature</publisher-name>), <fpage>155</fpage>&#x2013;<lpage>176</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Baumann</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kennington</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hough</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schlangen</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Recognising conversational speech: what an incremental ASR should do for a dialogue system and how to get there</article-title>, Editors <person-group person-group-type="editor">
<name>
<surname>Jokinen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wilcock</surname>
<given-names>G.</given-names>
</name>
</person-group> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Singapore</publisher-name>), <volume>427</volume>, <fpage>421</fpage>&#x2013;<lpage>432</lpage>. <pub-id pub-id-type="doi">10.1007/978-981-10-2585-3_35</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ben-Youssef</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Clavel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Essid</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bilac</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chamoux</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <source>UE HRI: a new dataset for the study of user engagement in spontaneous human robot interactions, in Icmi 2017 Proceedings of the 19th ACM international Conference on multimodal interaction</source>, <fpage>464</fpage>&#x2013;<lpage>472</lpage>. <pub-id pub-id-type="doi">10.1145/3136755.3136814</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bilac</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chamoux</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Gaze and filled pause detection for smooth human-robot conversations</article-title>,&#x201d; in <source>2017 IEEE-RAS 17th international conference on humanoid robotics (humanoids)</source>, <fpage>297</fpage>&#x2013;<lpage>304</lpage>. <pub-id pub-id-type="doi">10.1109/HUMANOIDS.2017.8246889</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>B&#xf6;gels</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Torreira</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Listeners use intonational phrase boundaries to project turn ends in spoken interaction</article-title>. <source>J. Phonetics</source> <volume>52</volume>, <fpage>46</fpage>&#x2013;<lpage>57</lpage>. <pub-id pub-id-type="doi">10.1016/j.wocn.2015.04.004</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bullock</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bredin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Garcia-Perera</surname>
<given-names>L. P.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Overlap-aware diarization: resegmentation using neural end-to-end overlapped speech detection</article-title>,&#x201d; in <source>Icassp 2020 - 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP)</source>, <fpage>7114</fpage>&#x2013;<lpage>7118</lpage>. <pub-id pub-id-type="doi">10.1109/ICASSP40776.2020.9053096</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Button</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1990</year>). &#x201c;<article-title>Going up a blind alley</article-title>,&#x201d; in <source>Computers and conversation</source> (<publisher-name>Elsevier</publisher-name>), <fpage>67</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-08-050264-9.50009-9</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Button</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sharrock</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>1995</year>). &#x201c;<article-title>On simulacrums of conversation: toward a clarification of the relevance of conversation analysis for human-computer interaction</article-title>,&#x201d; in <source>The social and interactional dimensions of human-computer interfaces</source> (<publisher-loc>New York, NY, US</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>), <fpage>107</fpage>&#x2013;<lpage>125</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A scope sensitive and result attentive model for multi-intent spoken language understanding</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>37</volume>, <fpage>12691</fpage>&#x2013;<lpage>12699</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v37i11.26493</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hua</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Toward high quality multi-object tracking and segmentation without mask supervision</article-title>. <source>IEEE Trans. Image Process.</source> <volume>33</volume>, <fpage>3369</fpage>&#x2013;<lpage>3384</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2024.3403497</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clark</surname>
<given-names>H. H.</given-names>
</name>
<name>
<surname>Wilkes-Gibbs</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Referring as a collaborative process</article-title>. <source>Cognition</source> <volume>22</volume>, <fpage>1</fpage>&#x2013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1016/0010-0277(86)90010-7</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Degutyte</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Astell</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The role of eye gaze in regulating turn taking in conversations: a systematized review of methods and findings</article-title>. <source>Front. Psychol.</source> <volume>12</volume>, <fpage>616471</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2021.616471</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dourish</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>Implications for design</article-title>,&#x201d; in <source>Conference proceedings/CHI 2006, conference on human factors in computing systems</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Grinter</surname>
<given-names>R.</given-names>
</name>
</person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>ACM Press</publisher-name>), <fpage>541</fpage>&#x2013;<lpage>550</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dourish</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Button</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>On technomethodology: foundational relationships between ethnomethodology and system design</article-title>. <source>Human-Computer Interact.</source> <volume>13</volume>, <fpage>395</fpage>&#x2013;<lpage>432</lpage>. <pub-id pub-id-type="doi">10.1207/s15327051hci1304_2</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Drew</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1992</year>). &#x201c;<article-title>Contested evidence in courtroom cross-examination: the case of a trial for rape</article-title>,&#x201d; in <source>Talk at work: interaction in institutional settings</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Drew</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Heritage</surname>
<given-names>J.</given-names>
</name>
</person-group> (<publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>), <fpage>470</fpage>&#x2013;<lpage>520</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Drew</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Couper-Kuhlen</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Requesting in social interaction</source>. <publisher-name>John Benjamins Publishing Company</publisher-name>. <pub-id pub-id-type="doi">10.1017/CBO9781107415324.004</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Driess</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sajjadi</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Lynch</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chowdhery</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ichter</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>Palm-e: an embodied multimodal language model</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>8469</fpage>&#x2013;<lpage>8488</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duncan</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1972</year>). <article-title>Some signals and rules for taking speaking turns in conversations</article-title>. <source>J. Personality Soc. Psychol.</source> <volume>23</volume>, <fpage>283</fpage>&#x2013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.1037/h0033031</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Duran</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Conversation analysis for computational modelling of task-oriented dialogue</source>. <publisher-loc>Bristol, England</publisher-loc>: <publisher-name>University of the West of England</publisher-name>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Egbert</surname>
<given-names>M. M.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Schisming: the collaborative transformation from a single conversation to multiple conversations</article-title>. <source>Res. Lang. Soc. Interact.</source> <volume>30</volume>, <fpage>1</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1207/s15327973rlsi3001_1</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ekstedt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Skantze</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>TurnGPT: a transformer-based language model for predicting turn-taking in spoken dialog</article-title>,&#x201d; in <source>Findings of the association for computational linguistics: emnlp 2020</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Cohn</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<publisher-name>Online: Association for Computational Linguistics</publisher-name>), <fpage>2981</fpage>&#x2013;<lpage>2990</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2020</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ekstedt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Skantze</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Projection of turn completion in incremental spoken dialogue systems</article-title>,&#x201d; in <source>Proceedings of the 22nd annual meeting of the special interest group on discourse and dialogue</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Levow</surname>
<given-names>G.-A.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sisman</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Singapore and Online: Association for Computational Linguistics</publisher-name>), <fpage>431</fpage>&#x2013;<lpage>437</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ekstedt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Skantze</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>How much does prosody help turn-taking? Investigations using voice activity projection models</article-title>,&#x201d; in <source>Proceedings of the 23rd annual meeting of the special interest group on discourse and dialogue</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Lemon</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Hakkani-Tur</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Ashrafzadeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Garcia</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Alikhani</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<publisher-loc>Edinburgh, UK</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>541</fpage>&#x2013;<lpage>551</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2022.sigdial-1.51</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Enfield</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Sidnell</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Intersubjectivity is activity plus accountability</article-title>,&#x201d; in <source>Oxford handbook of human symbolic evolution</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Nathalie Gontier</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Sinha</surname>
<given-names>C.</given-names>
</name>
</person-group> (<publisher-loc>Oxford, New York</publisher-loc>: <publisher-name>Oxford University Press</publisher-name>), <fpage>259</fpage>&#x2013;<lpage>288</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Fischer</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Reeves</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Porcheron</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sikveland</surname>
<given-names>R. O.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Progressivity for voice interface design</article-title>,&#x201d; in <source>Proceedings of the 1st international conference on conversational user interfaces</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1145/3342775.3342788</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ford</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Thompson</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1996</year>). &#x201c;<article-title>Interactional units in conversation: syntactic, intonational, and pragmatic resources for the management of turns</article-title>,&#x201d; in <source>Interaction and grammar</source> (<publisher-name>Cambridge University Press</publisher-name>), <fpage>134</fpage>&#x2013;<lpage>184</lpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ford</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Fox</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Thompson</surname>
<given-names>S. A.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Practices in the construction of turns</article-title>. <source>Pragmat. Q. Publ. Int. Pragmat. Assoc. (IPrA)</source> <volume>6</volume>, <fpage>427</fpage>&#x2013;<lpage>454</lpage>. <pub-id pub-id-type="doi">10.1075/prag.6.3.07for</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gangadharaiah</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Narayanaswamy</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Joint multiple intent detection and slot labeling for goal-oriented dialog</article-title>,&#x201d; in <source>Proceedings of the 2019 conference of the north American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Burstein</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Doran</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Solorio</surname>
<given-names>T.</given-names>
</name>
</person-group> (<publisher-loc>Minneapolis, Minnesota</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>564</fpage>&#x2013;<lpage>569</lpage>. <pub-id pub-id-type="doi">10.18653/v1/N19-1055</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Garfinkel</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sacks</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>1970</year>). <source>On formal structures of practical action, in Theoretical sociology: Perspectives and developments</source>, <fpage>337</fpage>&#x2013;<lpage>366</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gehle</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pitsch</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Dankert</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wrede</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Trouble-based group dynamics in real-world HRI: reactions on unexpected next moves of a museum guide robot</article-title>,&#x201d; in <source>2015 24th IEEE international Symposium on Robot and human interactive communication (RO-MAN)</source>, <fpage>407</fpage>&#x2013;<lpage>412</lpage>. <pub-id pub-id-type="doi">10.1109/ROMAN.2015.7333574</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ghosh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Do users need human-like conversational agents? &#x2013; Exploring conversational system design using framework of human needs</article-title>,&#x201d; in <source>Desires 2021 &#x2013; 2nd international conference on design of experimental search information REtrieval systems padua, Italy</source>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Goodwin</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>1981</year>). <source>Conversational Organization: interaction between speakers and hearers</source>. <publisher-loc>London and New York</publisher-loc>: <publisher-name>Academic Press</publisher-name>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gravano</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hirschberg</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Turn-taking cues in task-oriented dialogue</article-title>. <source>Comput. Speech &#x26; Lang.</source> <volume>25</volume>, <fpage>601</fpage>&#x2013;<lpage>634</lpage>. <pub-id pub-id-type="doi">10.1016/j.csl.2010.10.003</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gu</surname>
<given-names>J.-C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>Z.-H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>Y.-P.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Deep contextualized utterance representations for response selection and dialogue analysis</article-title>. <source>IEEE/ACM Trans. Audio, Speech, Lang. Process.</source> <volume>29</volume>, <fpage>2443</fpage>&#x2013;<lpage>2455</lpage>. <pub-id pub-id-type="doi">10.1109/TASLP.2021.3074788</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ham</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.-G.</given-names>
</name>
<name>
<surname>Jang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>K.-E.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>End-to-End neural pipeline for goal-oriented dialogue systems using GPT-2</article-title>,&#x201d; in <source>Proceedings of the 58th annual meeting of the association for computational linguistics</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Jurafsky</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schluter</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tetreault</surname>
<given-names>J.</given-names>
</name>
</person-group> (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Online: Association for Computational Linguistics</publisher-name>), <fpage>583</fpage>&#x2013;<lpage>592</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.54</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harvey Sacks</surname>
<given-names>E. A. S.</given-names>
</name>
<name>
<surname>Jefferson</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1974</year>). <article-title>A simplest systematics for the organization of turn-taking for conversation</article-title>. <source>Language</source> <volume>50</volume>, <fpage>696</fpage>&#x2013;<lpage>735</lpage>. <pub-id pub-id-type="doi">10.1353/lan.1974.0010</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Heeman</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kun</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Shyrokov</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Conventions in human-human multi-threaded dialogues: IUI 05 - 2005 international conference on intelligent user interfaces</article-title>,&#x201d; in <source>Proceedings of the 10th international conference on Intelligent user interfaces</source>, <fpage>293</fpage>&#x2013;<lpage>295</lpage>. <pub-id pub-id-type="doi">10.1145/1040830.1040903</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Heritage</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1984</year>). &#x201c;<article-title>A change-of-state token and its aspects of its sequential placement</article-title>,&#x201d; in <source>Structures of social action</source> (<publisher-name>Cambridge University Press</publisher-name>), <fpage>299</fpage>&#x2013;<lpage>345</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Heritage</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1998</year>). &#x201c;<article-title>Conversation analysis and institutional talk: analyzing distinctive turn-taking systems</article-title>,&#x201d; in <source>Proceedings of the 6th international congresss of IADA</source>, <fpage>3</fpage>&#x2013;<lpage>17</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Heritage</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Watson</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1979</year>). &#x201c;<article-title>Formulations as conversational objects</article-title>,&#x201d; in <source>Everyday Language: studies in ethnomethodology</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Psathas</surname>
<given-names>G.</given-names>
</name>
</person-group> (<publisher-loc>New York</publisher-loc>: <publisher-name>Irvington: Irvington</publisher-name>), <fpage>123</fpage>&#x2013;<lpage>162</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Singhal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>Language is not all you need: aligning perception with language models</article-title>,&#x201d; in <conf-name>Advances in neural information processing systems</conf-name>, <conf-loc>New Orleans, LA</conf-loc>. Editors <person-group person-group-type="editor">
<name>
<surname>Oh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Naumann</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Globerson</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Saenko</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hardt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <volume>36</volume>, <fpage>72096</fpage>&#x2013;<lpage>72109</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Imrattanatrai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Fukuda</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>End-to-End task-oriented dialogue systems based on schema</article-title>,&#x201d; in <source>Findings of the association for computational linguistics: acl 2023</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Rogers</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Boyd-Graber</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Okazaki</surname>
<given-names>N.</given-names>
</name>
</person-group> (<publisher-loc>Toronto, Canada</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>10148</fpage>&#x2013;<lpage>10161</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2023</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ishii</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Otsuka</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kumano</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yamato</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Analysis of respiration for prediction of who will Be next speaker and when? In multi-party meetings</article-title>,&#x201d; in <source>Proceedings of the 16th international conference on multimodal interaction</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>18</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1145/2663204.2663271</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Janarthanam</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lemon</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Multi-threaded interaction management for dynamic spatial applications</article-title>,&#x201d; in <source>Proceedings of the EACL 2014 workshop on dialogue in motion</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Dalmas</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>G&#xf6;tze</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gustafson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Janarthanam</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kleindienst</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mueller</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<publisher-loc>Gothenburg, Sweden</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>48</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.3115/v1/W14-0208</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jefferson</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2004</year>). &#x201c;<article-title>A sketch of some orderly aspects of overlap in natural conversation (1975)</article-title>,&#x201d; in <source>Conversation analysis: studies from the first generation</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Lerner</surname>
<given-names>G. H.</given-names>
</name>
</person-group> (<publisher-loc>Amsterdam/Philadelphia</publisher-loc>: <publisher-name>John Benjamins Publishing Company</publisher-name>), <fpage>13</fpage>&#x2013;<lpage>23</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>H. T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A survey of human action analysis in hri applications</article-title>. <source>IEEE Trans. Circuits Syst. Video Technol.</source> <volume>30</volume>, <fpage>2114</fpage>&#x2013;<lpage>2128</lpage>. <pub-id pub-id-type="doi">10.1109/TCSVT.2019.2912988</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jokinen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Furukawa</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nishida</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yamamoto</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Gaze and turn-taking behavior in casual conversational interactions</article-title>. <source>ACM Trans. Interact. Intell. Syst.</source>, <volume>3</volume>, <fpage>1</fpage>. <lpage>30</lpage>. <pub-id pub-id-type="doi">10.1145/2499474.2499481</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kassner</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tafjord</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Sch&#xfc;tze</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>BeliefBank: adding memory to a pre-trained language model for a systematic notion of belief</article-title>,&#x201d; in <source>Proceedings of the 2021 conference on empirical methods in natural language processing</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Moens</surname>
<given-names>M.-F.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Specia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yih</surname>
<given-names>S. W.-t.</given-names>
</name>
</person-group> (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>8849</fpage>&#x2013;<lpage>8861</lpage>. <comment>(Online and Punta Cana, Dominican Republic</comment>. <pub-id pub-id-type="doi">10.18653/v1/2021.emnlp-main.697</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kendon</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1967</year>). <article-title>Some functions of gaze-direction in social interaction</article-title>. <source>Acta Psychol.</source> <volume>26</volume>, <fpage>22</fpage>&#x2013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1016/0001-6918(67)90005-4</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kendrick</surname>
<given-names>K. H.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Dingemanse</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Floyd</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gipper</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hayano</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Sequence organization: a universal infrastructure for social action</article-title>. <source>J. Pragmat.</source> <volume>168</volume>, <fpage>119</fpage>&#x2013;<lpage>138</lpage>. <pub-id pub-id-type="doi">10.1016/j.pragma.2020.06.009</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kendrick</surname>
<given-names>K. H.</given-names>
</name>
<name>
<surname>Drew</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2014a</year>). &#x201c;<article-title>The putative preference for offers over requests</article-title>,&#x201d; in <source>Requesting in social interaction</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Drew</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Couper-Kuhlen</surname>
<given-names>E.</given-names>
</name>
</person-group> (<publisher-name>John Benjamins Publishing Company), Studies in Language and Social Interaction</publisher-name>), <fpage>87</fpage>&#x2013;<lpage>114</lpage>. <pub-id pub-id-type="doi">10.1075/slsi.26.04ken</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kharitonov</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Polyak</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Adi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Copet</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lakhotia</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Text-free prosody-aware generative spoken language modeling</article-title>. <source>ACL 2022-Association Comput. Linguistics</source> <volume>1</volume>, <fpage>8666</fpage>&#x2013;<lpage>8681</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2109.03264</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ryu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G. G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Two-stage multi-intent detection for spoken language understanding</article-title>. <source>Multimedia Tools Appl.</source> <volume>76</volume>, <fpage>11377</fpage>&#x2013;<lpage>11390</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-016-3724-4</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kl&#xfc;wer</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). <source>
<italic>Social talk Capabilities for dialogue systems</italic> (publisher&#x3d;Saarl&#xe4;ndische universit&#xe4;ts-und landesbibliothek)</source>.</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kummerfeld</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Gouravajhala</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Peper</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Athreya</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Gunasekara</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ganhotra</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>A large-scale corpus for conversation disentanglement</article-title>,&#x201d; in <source>Proceedings of the 57th annual meeting of the association for computational linguistics</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Korhonen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Traum</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>M&#xe0;rquez</surname>
<given-names>L.</given-names>
</name>
</person-group> (<publisher-loc>Florence, Italy</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>3846</fpage>&#x2013;<lpage>3856</lpage>. <pub-id pub-id-type="doi">10.18653/v1/P19-1374</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kunneman</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hindriks</surname>
<given-names>K. V.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>A sequence-based dialog management framework for Co-regulated dialog</article-title>,&#x201d; in <source>HHAI2022: augmenting human intellect</source> (<publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>IOS Press</publisher-name>), <fpage>143</fpage>&#x2013;<lpage>156</lpage>. <pub-id pub-id-type="doi">10.3233/FAIA220195</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Landesberger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ehrlich</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Investigating strategies for resolving misunderstood utterances with multiple intents</article-title>,&#x201d; in <conf-name>Proceedings of the 22nd workshop on the semantics and pragmatics of dialogue (AixDial)</conf-name>, <conf-loc>Aix-en-Provence, France</conf-loc> (<publisher-name>Online</publisher-name>).</citation>
</ref>
<ref id="B61">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Landesberger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ehrlich</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Towards finding appropriate responses to multi-intents - SPM: sequential prioritisation model</article-title>,&#x201d; in <source>Proceedings of the 23rd workshop on the semantics and pragmatics of dialogue - poster abstracts</source>, <fpage>248</fpage>.</citation>
</ref>
<ref id="B62">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Landesberger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ehrlich</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Detecting urgency in speech with personalised acoustic features</article-title>,&#x201d; in <source>Proceedings of the 24th workshop on the semantics and pragmatics of dialogue - short papers</source>, <fpage>248</fpage>&#x2013;<lpage>250</lpage>.</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Tanaka</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Affiliation and alignment in responding actions</article-title>. <source>J. Pragmat.</source> <volume>100</volume>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1016/j.pragma.2016.05.008</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lemon</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Gruenstein</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Battle</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2002</year>). &#x201c;<article-title>Multi-tasking and collaborative activities in dialogue systems</article-title>,&#x201d; in <source>Proceedings of the third SIGdial workshop on discourse and dialogue</source> (<publisher-loc>Philadelphia, Pennsylvania, USA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>113</fpage>&#x2013;<lpage>124</lpage>. <pub-id pub-id-type="doi">10.3115/1118121.1118137</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Levinson</surname>
<given-names>S. C.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>&#x201c;On the human interaction engine&#x201d;</article-title>,&#x201d; in <source>Roots of human sociality</source> (<publisher-loc>London, Berg</publisher-loc>: <publisher-name>Routledge</publisher-name>), <fpage>39</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1023/a:1018829907604</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Levinson</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Torreira</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Timing in turn-taking and its implications for processing models of language</article-title>. <source>Front. Psychol.</source> <volume>6</volume>, <fpage>731</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2015.00731</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Fei</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <source>Revisiting conversation discourse for dialogue disentanglement</source>.</citation>
</ref>
<ref id="B68">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>A light weight model for active speaker detection</article-title>,&#x201d; in <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>, <fpage>22932</fpage>&#x2013;<lpage>22941</lpage>.</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Licoppe</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rollet</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Je dois y aller Analyses de s&#xe9;quences de cl&#xf4;tures entre humains et robot</article-title>. <source>R&#xe9;seaux</source> <volume>2020/2</volume>, <fpage>151</fpage>&#x2013;<lpage>193</lpage>. <pub-id pub-id-type="doi">10.3917/res.220.0151</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Y. T.</given-names>
</name>
<name>
<surname>Papangelis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hakkani-Tur</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Knowledge-grounded conversational data augmentation with generative conversational networks</article-title>,&#x201d; in <source>Proceedings of the 23rd annual meeting of the special interest group on discourse and dialogue</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Lemon</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Hakkani-Tur</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Ashrafzadeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Garcia</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Alikhani</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<publisher-loc>Edinburgh, UK</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>26</fpage>&#x2013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2022.sigdial-1.3</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lindstr&#xf6;m</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Language as social action: a study of how senior citizens request assistance with practical tasks in the Swedish home help service</article-title>,&#x201d; in <source>Syntax and lexis in conversation: studies on the use of linguistic resources in talk-in-interaction</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Hakulinen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Selting</surname>
<given-names>M.</given-names>
</name>
</person-group> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>Benjamins</publisher-name>), <fpage>209</fpage>&#x2013;<lpage>230</lpage>.</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.-C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>End-to-end transition-based online dialogue disentanglement</article-title>. <source>Proc. Twenty-Ninth Int. Jt. Conf. Artif. Intell.</source> <volume>20</volume>, <fpage>3868</fpage>&#x2013;<lpage>3874</lpage>. <pub-id pub-id-type="doi">10.24963/ijcai.2020/535</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C.-T.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021b</year>). <article-title>Fuzzy detection aided real-time and robust visual tracking under complex environments</article-title>. <source>IEEE Trans. Fuzzy Syst.</source> <volume>29</volume>, <fpage>90</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1109/TFUZZ.2020.3006520</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lohse</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hanheide</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pitsch</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rohlfing</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Sagerer</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Improving HRI design by applying systemic interaction analysis (SInA)</article-title>. <source>Interact. Stud.</source> <volume>10</volume>, <fpage>298</fpage>&#x2013;<lpage>323</lpage>. <pub-id pub-id-type="doi">10.1075/is.10.3.03loh</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lund</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Basso-Fossali</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mazur</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ollagnier-Beldame</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rabatel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bond&#xec;</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <source>
<italic>Language is a complex adaptive system</italic> (Language Science Press)</source>. <publisher-loc>Berlin</publisher-loc>: <publisher-name>Language Science Press</publisher-name>. <pub-id pub-id-type="doi">10.5281/zenodo.6546419</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Majlesi</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Cumbal</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Engwall</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Gillet</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kunitz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lymer</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Managing turn-taking in human-robot interactions: the case of projections and overlaps, and the anticipation of turn design by human participants</article-title>. <source>Soc. Interact. Video-Based Stud. Hum. Sociality</source> <volume>6</volume>. <comment>Number: 1</comment>. <pub-id pub-id-type="doi">10.7146/si.v6i1.137380</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maraev</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Bernardy</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Ginzburg</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Dialogue management with linear logic: the role of metavariables in questions and clarifications</article-title>. <source>Trait. Autom. Des. Langues</source> <volume>61</volume>, <fpage>43</fpage>&#x2013;<lpage>67</lpage>.</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marvasti-Zadeh</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ghanei-Yakhdan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kasaei</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep learning for visual tracking: a comprehensive survey</article-title>. <source>IEEE Trans. Intelligent Transp. Syst.</source> <volume>23</volume>, <fpage>3943</fpage>&#x2013;<lpage>3968</lpage>. <pub-id pub-id-type="doi">10.1109/tits.2020.3046478</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meena</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Skantze</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gustafson</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Data-driven models for timing feedback responses in a Map Task dialogue system</article-title>. <source>Comput. Speech &#x26; Lang.</source> <volume>28</volume>, <fpage>903</fpage>&#x2013;<lpage>922</lpage>. <pub-id pub-id-type="doi">10.1016/j.csl.2014.02.002</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Min</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tripathi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guha</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Majumdar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Learning long-term spatial-temporal graphs for active speaker detection</article-title>,&#x201d; in <source>European conference on computer vision</source> (<publisher-name>Springer</publisher-name>), <fpage>371</fpage>&#x2013;<lpage>387</lpage>.</citation>
</ref>
<ref id="B81">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Moore</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Arar</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Conversational ux design: a practitioner&#x2019;s Guide to the natural conversation framework (morgan and claypool)</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>.</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muhle</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Robots as addressable non-persons: an analysis of categorial work at the boundaries of the social world</article-title>. <source>Front. Sociol.</source> <volume>9</volume>, <fpage>1260823</fpage>. <pub-id pub-id-type="doi">10.3389/fsoc.2024.1260823</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nakano</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Komatani</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A framework for building closed-domain chat dialogue systems</article-title>. <source>Knowledge-Based Syst.</source> <volume>204</volume>, <fpage>106212</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2020.106212</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Natarajan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chhipa</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yadav</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gogoi</surname>
<given-names>D. V.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Unified multi intent order and slot prediction using selective learning propagation</article-title>,&#x201d; in <source>Proceedings of the workshop on joint NLP modelling for conversational AI @ ICON 2020</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Mukherjee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Samal</surname>
<given-names>R.</given-names>
</name>
</person-group> (<publisher-loc>Patna, India</publisher-loc>: <publisher-name>NLP Association of India NLPAI</publisher-name>), <fpage>10</fpage>&#x2013;<lpage>18</lpage>.</citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nazir</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Lebrun</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Improving the acceptability of social robots: make them look different from humans</article-title>. <source>PLOS ONE</source> <volume>18</volume>, <fpage>e0287507</fpage>. <comment>Publisher: Public Library of Science</comment>. <pub-id pub-id-type="doi">10.1371/journal.pone.0287507</pub-id>
</citation>
</ref>
<ref id="B86">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kishan</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chadha</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Efficient fine-tuning large language models for knowledge-aware response planning</article-title>,&#x201d; in <source>Machine learning and knowledge discovery in databases: research track</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Koutra</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Plant</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gomez Rodriguez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Baralis</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bonchi</surname>
<given-names>F.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>), <fpage>593</fpage>&#x2013;<lpage>611</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-43415-0_35</pub-id>
</citation>
</ref>
<ref id="B87">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Papaioannou</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Dondrup</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lemon</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Human-robot interaction requires more than slot filling - multi-threaded dialogue for collaborative tasks and social conversation</article-title>,&#x201d; in <source>Proceedings of the FAIM/ISCA workshop on artificial intelligence for multimodal human robot interaction</source>, <fpage>61</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.21437/AI-MHRI.2018-15</pub-id>
</citation>
</ref>
<ref id="B88">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pelikan</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Broth</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Why that nao? how humans adapt to a conventional humanoid robot in taking turns-at-talk</article-title>. <source>Proc. 2016 CHI Conf. Hum. Factors Comput. Syst.</source> <volume>16</volume>, <fpage>4921</fpage>&#x2013;<lpage>4932</lpage>. <pub-id pub-id-type="doi">10.1145/2858036.2858478</pub-id>
</citation>
</ref>
<ref id="B89">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pelikan</surname>
<given-names>H. R. M.</given-names>
</name>
<name>
<surname>Reeves</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cantarutti</surname>
<given-names>M. N.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Whose perspective are we studying in ethnographic HRI?</article-title>,&#x201d; in <source>Ethnography for HRI: embodied, embedded, messy and everyday</source> (<publisher-loc>Boulder, CO</publisher-loc>: <publisher-name>Workshop at HRI&#x2019;24</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B90">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pitsch</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Limits and opportunities for mathematizing communicational conduct for social robotics in the real world? Toward enabling a robot to make use of the human&#x2019;s competences</article-title>. <source>AI &#x26; Soc.</source> <volume>31</volume>, <fpage>587</fpage>&#x2013;<lpage>593</lpage>. <pub-id pub-id-type="doi">10.1007/s00146-015-0629-0</pub-id>
</citation>
</ref>
<ref id="B91">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pitsch</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Vollmer</surname>
<given-names>A.-L.</given-names>
</name>
<name>
<surname>M&#xfc;hlig</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Robot feedback shapes the tutor&#x2019;s presentation: how a robot&#x2019;s online gaze strategies lead to micro-adaptation of the human&#x2019;s conduct</article-title>. <source>Interact. Stud. Soc. Behav. Commun. Biol. Artif. Syst.</source> <volume>14</volume>, <fpage>268</fpage>&#x2013;<lpage>296</lpage>. <pub-id pub-id-type="doi">10.1075/is.14.2.06pit</pub-id>
</citation>
</ref>
<ref id="B92">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pomerantz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heritage</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Preference</article-title>,&#x201d; in <source>The handbook of conversation analysis</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Sidnell</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stivers</surname>
<given-names>T.</given-names>
</name>
</person-group> (<publisher-name>Wiley-Blackwell</publisher-name>), <fpage>210</fpage>&#x2013;<lpage>228</lpage>.</citation>
</ref>
<ref id="B93">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Porcheron</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Sharples</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Do animals have accents? talking with agents in multi-party conversation</article-title>,&#x201d; in <source>Proceedings of the 2017 ACM conference on computer supported cooperative work and social computing</source>, <fpage>207</fpage>&#x2013;<lpage>219</lpage>.</citation>
</ref>
<ref id="B94">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Che</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>AGIF: an adaptive graph-interactive framework for joint multiple intent detection and slot filling</article-title>. <source>ArXiv:2004.10087</source>. <comment>[cs, eess]</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2004.10087</pub-id>
</citation>
</ref>
<ref id="B95">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wenjing</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhangli</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>B&#x26;net: combining bidirectional LSTM and self-attention for end-to-end learning of task-oriented dialogue system</article-title>. <source>Speech Commun.</source> <volume>125</volume>, <fpage>15</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1016/j.specom.2020.09.005</pub-id>
</citation>
</ref>
<ref id="B96">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reeves</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Porcheron</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Conversational ai: respecifying participation as regulation</article-title>,&#x201d; in <source>The SAGE handbook of digital society</source> (<publisher-loc>London</publisher-loc>: <publisher-name>SAGE Publications Ltd</publisher-name>), <fpage>573</fpage>&#x2013;<lpage>592</lpage>. <pub-id pub-id-type="doi">10.4135/9781529783193.n32</pub-id>
</citation>
</ref>
<ref id="B97">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reeves</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Porcheron</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>This is not what we wanted&#x2019;: designing for conversation with voice interfaces</article-title>. <source>ACM Interact.</source> <volume>26</volume>, <fpage>46</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1145/3296699</pub-id>
</citation>
</ref>
<ref id="B98">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rochet-Capellan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fuchs</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Take a breath and take the turn: how breathing meets turns in spontaneous dialogue</article-title>. <source>Philosophical Trans. R. Soc. B Biol. Sci.</source> <volume>369</volume>, <fpage>20130399</fpage>. <pub-id pub-id-type="doi">10.1098/rstb.2013.0399</pub-id>
</citation>
</ref>
<ref id="B99">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ros&#xe9;</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Di Eugenio</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Levin</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Van Ess-Dykema</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>1995</year>). &#x201c;<article-title>Discourse processing of dialogues with multiple threads</article-title>,&#x201d; in <source>33rd annual meeting of the association for computational linguistics</source> (<publisher-loc>Cambridge, Massachusetts, USA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>31</fpage>&#x2013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.3115/981658.981663</pub-id>
</citation>
</ref>
<ref id="B100">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruiter</surname>
<given-names>J.-P. d.</given-names>
</name>
<name>
<surname>Mitterer</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Enfield</surname>
<given-names>N. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Projecting the end of a speaker&#x2019;s turn: a cognitive cornerstone of conversation</article-title>. <source>Language</source> <volume>82</volume>, <fpage>515</fpage>&#x2013;<lpage>535</lpage>. <pub-id pub-id-type="doi">10.1353/lan.2006.0130</pub-id>
</citation>
</ref>
<ref id="B101">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schegloff</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>1968</year>). <article-title>Sequencing in conversational openings</article-title>. <source>Dir. sociolinguistics</source> <volume>70</volume>, <fpage>1075</fpage>&#x2013;<lpage>1095</lpage>. <pub-id pub-id-type="doi">10.1525/aa.1968.70.6.02a00030</pub-id>
</citation>
</ref>
<ref id="B102">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schegloff</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>1996</year>). &#x201c;<article-title>Issues of relevance for discourse analysis: contingency in action, interaction and Co-participant context</article-title>,&#x201d; in <source>Computational and conversational discourse</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Hovy</surname>
<given-names>E. H.</given-names>
</name>
<name>
<surname>Scott</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>3</fpage>&#x2013;<lpage>35</lpage>.</citation>
</ref>
<ref id="B103">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schegloff</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Body torque</article-title>. <source>Soc. Res.</source> <volume>65</volume>, <fpage>535</fpage>&#x2013;<lpage>596</lpage>.</citation>
</ref>
<ref id="B104">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schegloff</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>2002</year>). &#x201c;<article-title>Accounts of conduct in interaction: interruption, overlap and turn-taking</article-title>,&#x201d; in <source>Handbook of sociological theory</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Turner</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<publisher-loc>New York</publisher-loc>: <publisher-name>Plenum Press</publisher-name>), <fpage>287</fpage>&#x2013;<lpage>321</lpage>. <comment>chap. 15</comment>. <pub-id pub-id-type="doi">10.1007/0-387-36274-6_15</pub-id>
</citation>
</ref>
<ref id="B105">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schegloff</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>2007</year>). <source>Sequence organization in interaction: volume 1: a primer in conversation analysis</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation>
</ref>
<ref id="B106">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schegloff</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Sacks</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>1973</year>). <article-title>Opening up closings</article-title>. <source>Semiotica</source> <volume>8</volume>, <fpage>289</fpage>&#x2013;<lpage>327doi</lpage>. <pub-id pub-id-type="doi">10.1515/semi.1973.8.4.289</pub-id>
</citation>
</ref>
<ref id="B107">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Selting</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>The construction of units in conversational talk</article-title>. <source>Language</source> <volume>29</volume>, <fpage>477</fpage>&#x2013;<lpage>517</lpage>. <pub-id pub-id-type="doi">10.1017/s0047404500004012</pub-id>
</citation>
</ref>
<ref id="B108">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Sha</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>We know what you will ask: a dialogue system for multi-intent switch and prediction. In <italic>natural Language Processing and Chinese computing</italic>
</article-title>,&#x201d;. Editors <person-group person-group-type="editor">
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kan</surname>
<given-names>M.-Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zan</surname>
<given-names>H.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>93</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-32233-5_8</pub-id>
</citation>
</ref>
<ref id="B109">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Skantze</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Turn-taking in conversational systems and human-robot interaction: a review</article-title>. <source>Comput. Speech &#x26; Lang.</source> <volume>67</volume>, <fpage>101178</fpage>. <pub-id pub-id-type="doi">10.1016/j.csl.2020.101178</pub-id>
</citation>
</ref>
<ref id="B110">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.-N.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Generating persona consistent dialogues by exploiting natural language inference</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>34</volume>, <fpage>8878</fpage>&#x2013;<lpage>8885</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v34i05.6417</pub-id>
</citation>
</ref>
<ref id="B111">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Quangang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yubin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Enhancing joint multiple intent detection and slot filling with global intent-slot Co-occurrence</article-title>,&#x201d; in <source>Proceedings of the 2022 conference on empirical methods in natural language processing</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Goldberg</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kozareva</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<publisher-loc>Abu Dhabi, United Arab Emirates</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>7967</fpage>&#x2013;<lpage>7977</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main</pub-id>
</citation>
</ref>
<ref id="B112">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stivers</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>A preference for progressivity in interaction</article-title>. <source>Lang. Soc.</source> <volume>35</volume>, <fpage>367</fpage>&#x2013;<lpage>392</lpage>. <pub-id pub-id-type="doi">10.1017/S0047404506060179</pub-id>
</citation>
</ref>
<ref id="B113">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Stommel</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>De Rijk</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Boumans</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>&#x201c;Pepper, what do you mean?&#x201d; Miscommunication and repair in robot-led survey interaction</article-title>,&#x201d; in <source>
<italic>2022 31st IEEE international Conference on Robot and human interactive communication (RO-MAN)</italic> (napoli, Italy: ieee)</source>, <fpage>385</fpage>&#x2013;<lpage>392</lpage>. <pub-id pub-id-type="doi">10.1109/RO-MAN53752.2022.9900528</pub-id>
</citation>
</ref>
<ref id="B114">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Towards fewer hallucinations in knowledge-grounded dialogue generation via augmentative and contrastive knowledge-dialogue</article-title>,&#x201d; in <source>Proceedings of the 61st annual meeting of the association for computational linguistics (volume 2: short papers)</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Rogers</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Boyd-Graber</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Okazaki</surname>
<given-names>N.</given-names>
</name>
</person-group> (<publisher-loc>Toronto, Canada</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>1741</fpage>&#x2013;<lpage>1750</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2023.acl-short.148</pub-id>
</citation>
</ref>
<ref id="B115">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tisserand</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Armetta</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Baldauf-Quilliatre</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bouquin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hassas</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lefort</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Sequential annotations for naturally-occurring HRI: first insights</article-title>,&#x201d; in <source>Proceedings of workshop on human-robot conversational interaction (HRCI workshop &#x2019;23)</source> (<publisher-loc>Stockholm, SE</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="B116">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tisserand</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Baldauf-Quilliatre</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Rejecting a robot&#x2019;s offer: an analysis of preference</article-title>. <source>Discourse and communication</source>. <pub-id pub-id-type="doi">10.1177/17504813241271486</pub-id>
</citation>
</ref>
<ref id="B117">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tunser</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Collaborer et int&#xe9;ragir dans les bureaux: l&#x2019;&#xe9;mergence mat&#xe9;rielle, verbale et incarn&#xe9;e de l&#x2019;organisation</article-title>. <comment>PhD thesis</comment>. <publisher-loc>Paris</publisher-loc>: <publisher-name>ENST</publisher-name>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://theses.fr/2014ENST0030">https://theses.fr/2014ENST0030</ext-link>
</comment>.</citation>
</ref>
<ref id="B118">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tuyen</surname>
<given-names>N. T. V.</given-names>
</name>
<name>
<surname>Georgescu</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Di Giulio</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Celiktutan</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>A multimodal dataset for robot learning to imitate social human-human interaction</article-title>,&#x201d; in <source>Companion of the 2023 ACM/IEEE international conference on human-robot interaction</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>238</fpage>&#x2013;<lpage>242</lpage>. <pub-id pub-id-type="doi">10.1145/3568294.3580080</pub-id>
</citation>
</ref>
<ref id="B119">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Velkovska</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zouinar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Veyrier</surname>
<given-names>C.-A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Les relations aux machines conversationnelles: Vivre avec les assistants vocaux &#xe0; la maison</article-title>. <source>R&#xe9;seaux N&#xb0;220-221</source> <volume>N&#xb0; 220-221</volume>, <fpage>47</fpage>&#x2013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.3917/res.220.0047</pub-id>
</citation>
</ref>
<ref id="B120">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>A survey on large language model based autonomous agents</article-title>. <source>Front. Comput. Sci.</source> <volume>18</volume>, <fpage>186345</fpage>. <pub-id pub-id-type="doi">10.1007/s11704-024-40231-1</pub-id>
</citation>
</ref>
<ref id="B121">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Webb</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2000</year>). &#x201c;<article-title>Rule-based dialogue management systems</article-title>,&#x201d; in <source>Proceedings of the 3rd international workshop on human-computer conversation</source> (<publisher-loc>Bellagio, Italy</publisher-loc>), <fpage>3</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B122">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wen</surname>
<given-names>T.-H.</given-names>
</name>
<name>
<surname>Vandyke</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mrk&#x161;i&#x107;</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ga&#x161;i&#x107;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rojas-Barahona</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>P.-H.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>A network-based end-to-end trainable task-oriented dialogue system</article-title>,&#x201d; in <source>Proceedings of the 15th conference of the European chapter of the association for computational linguistics: volume 1, long papers</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Lapata</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Blunsom</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Koller</surname>
<given-names>A.</given-names>
</name>
</person-group> (<publisher-loc>Valencia, Spain</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>438</fpage>&#x2013;<lpage>449</lpage>.</citation>
</ref>
<ref id="B123">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wigdor</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>de Greeff</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Looije</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Neerincx</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>How to improve human-robot interaction with Conversational Fillers</article-title>,&#x201d; in <source>2016 25th IEEE international symposium on robot and human interactive communication</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>RO-MAN</publisher-name>), <fpage>219</fpage>&#x2013;<lpage>224</lpage>. <pub-id pub-id-type="doi">10.1109/ROMAN.2016.7745134</pub-id>
</citation>
</ref>
<ref id="B124">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>T.-W.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Juang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>A label-aware BERT attention network for zero-shot multi-intent detection in spoken language understanding</article-title>,&#x201d; in <conf-name>Proceedings of the 2021 conference on empirical methods in natural language processing</conf-name>, <conf-loc>Online and Punta Cana, Dominican Republic</conf-loc>. Editors <person-group person-group-type="editor">
<name>
<surname>Moens</surname>
<given-names>M.-F.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Specia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yih</surname>
<given-names>S. W.-t.</given-names>
</name>
</person-group> (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>4884</fpage>&#x2013;<lpage>4896</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2021.emnlp-main.399</pub-id>
</citation>
</ref>
<ref id="B125">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Sarikaya</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Exploiting shared information for multi-intent natural language sentence classification</article-title>,&#x201d; in <source>
<italic>Interspeech 2013</italic> (ISCA)</source>, <fpage>3785</fpage>&#x2013;<lpage>3789</lpage>. <pub-id pub-id-type="doi">10.21437/Interspeech.2013-599</pub-id>
</citation>
</ref>
<ref id="B126">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yamazaki</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yoshikawa</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kawamoto</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mizumoto</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ohagi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sato</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Building a hospitable and reliable dialogue system for android robots: a scenario-based approach with large language models</article-title>. <source>Adv. Robot.</source> <volume>37</volume>, <fpage>1364</fpage>&#x2013;<lpage>1381</lpage>. <pub-id pub-id-type="doi">10.1080/01691864.2023.2244554</pub-id>
</citation>
</ref>
<ref id="B127">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Heeman</surname>
<given-names>P. A.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Context restoration in multi-tasking dialogue</article-title>,&#x201d; in <source>Proceedings of the 14th international conference on Intelligent user interfaces</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery), IUI &#x2019;09</publisher-name>), <fpage>373</fpage>&#x2013;<lpage>378</lpage>. <pub-id pub-id-type="doi">10.1145/1502650.1502703</pub-id>
</citation>
</ref>
<ref id="B128">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Heeman</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Kun</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). &#x201c;<article-title>Switching to real-time tasks in multi-tasking dialogue</article-title>,&#x201d; in <source>Proceedings of the 22nd international conference on computational linguistics (coling 2008)</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Scott</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>H.</given-names>
</name>
</person-group> (<publisher-loc>Manchester, UK</publisher-loc>: <publisher-name>Coling 2008 Organizing Committee</publisher-name>), <fpage>1025</fpage>&#x2013;<lpage>1032</lpage>.</citation>
</ref>
<ref id="B129">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Bytetrack: multi-object tracking by associating every detection box</article-title>,&#x201d; in <source>European conference on computer vision</source> (<publisher-name>Springer</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>21</lpage>.</citation>
</ref>
<ref id="B130">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lau</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Findings on conversation disentanglement</article-title>,&#x201d; in <source>Proceedings of the the 19th annual workshop of the australasian language technology association</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Rahimi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lane</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zuccon</surname>
<given-names>G.</given-names>
</name>
</person-group>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <comment>(Online: Australasian Language Technology Association)</comment>.</citation>
</ref>
</ref-list>
</back>
</article>