<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Commun.</journal-id>
<journal-title>Frontiers in Communication</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Commun.</abbrev-journal-title>
<issn pub-type="epub">2297-900X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcomm.2025.1524453</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Communication</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Gaze behaviour and vocal feedback in task-based dyadic conversations with and without eye contact</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Sbranna</surname> <given-names>Simona</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2889527/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Savino</surname> <given-names>Michelina</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2191474/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Baills</surname> <given-names>Florence</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1577982/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Grice</surname> <given-names>Martine</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/232419/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>IfL-Phonetics, University of Cologne</institution>, <addr-line>Cologne</addr-line>, <country>Germany</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Education, Psychology, Communication, University of Bari Aldo Moro</institution>, <addr-line>Bari</addr-line>, <country>Italy</country></aff>
<aff id="aff3"><sup>3</sup><institution>Faculty of Arts, Universitat de Lleida</institution>, <addr-line>Lleida</addr-line>, <country>Spain</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0003">
<p>Edited by: Maria Grazia Sindoni, University of Messina, Italy</p>
</fn>
<fn fn-type="edited-by" id="fn0004">
<p>Reviewed by: Isabella Poggi, Roma Tre University, Italy</p>
<p>Andy L&#x00FC;cking, Goethe University Frankfurt, Germany</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Simona Sbranna, <email>s.sbranna@outlook.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>05</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>10</volume>
<elocation-id>1524453</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>04</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Sbranna, Savino, Baills and Grice.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Sbranna, Savino, Baills and Grice</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Human communication is multimodal, with verbal and non-verbal cues such as eye gaze and vocal feedback being crucial for managing interactions. While much research has focused on eye gaze and turn alternation, few studies explore its relationship with turn-regulating vocal feedback. This study investigates this interplay during a Tangram game in Italian under two visibility conditions: face-to-face and separated by a screen. The results show that feedback producers rarely look at receivers, while receivers more frequently look at producers, suggesting that they might be eliciting vocal feedback. Without visual contact, gaze shifts decrease and vocal feedback increases. Interestingly, when visual contact is absent, gaze directed towards where the addressee is sitting does not coincide with vocal feedback, raising questions about what prompts this gaze.</p>
</abstract>
<kwd-group>
<kwd>turn-taking</kwd>
<kwd>eye gaze</kwd>
<kwd>vocal feedback</kwd>
<kwd>face-to-face dialogue</kwd>
<kwd>non-visibility</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="0"/>
<equation-count count="0"/>
<ref-count count="102"/>
<page-count count="16"/>
<word-count count="13961"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Multimodality of Communication</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>Human communication integrates several information channels, such as words, voice modulation (i.e., prosody), facial expressions, and gestures, enhancing the overall communicative impact in conversations (e.g., <xref ref-type="bibr" rid="ref72">Paulmann et al., 2009</xref>). This is particularly true in face-to-face interactions, where all these elements play a role in regulating social interaction and help to jointly create meaning (e.g., <xref ref-type="bibr" rid="ref74">Rasenberg et al., 2022</xref>). During conversation, interlocutors maintain the flow of their dialogue by speaking one at a time and taking turns in a cooperative manner. Interactions have been described as complex dynamic systems (<xref ref-type="bibr" rid="ref48">Hollenstein, 2013</xref>; <xref ref-type="bibr" rid="ref21">Cameron and Larsen-Freeman, 2007</xref>) since they emerge dynamically and evolve through a process of co-regulation between interlocutors (<xref ref-type="bibr" rid="ref49">Hu and Chen, 2021</xref> for a review; <xref ref-type="bibr" rid="ref93">Stahl, 2016</xref> for task-based dialogues). Turn-taking is believed to be driven by social signals consisting of verbal or non-verbal cues exchanged between individuals engaged in a conversation (e.g., <xref ref-type="bibr" rid="ref59">Kendrick et al., 2023</xref>). These signals serve as indicators of readiness to shift from speaking to listening or willingness to take the turn in speaking. Gaze cues (i.e., the way speakers orient their gaze) and vocal feedback (i.e., signals of active listening) have been shown to play important roles in the management of turn-taking in conversations. However, in relation to speakers&#x2019; intention of taking or not taking the floor, they have been traditionally investigated as separate entities (e.g., see <xref ref-type="bibr" rid="ref27">Degutyte and Astell, 2021</xref> for a review of studies on eye gaze and turn alternation, and <xref ref-type="bibr" rid="ref28">Drummond and Hopper, 1993</xref> as an example of a study on vocal feedback and turn alternation).</p>
<p>In the present study, we explore the relationship between turn-regulating vocal feedback and eye gaze behaviour in dyadic task-based conversations to shed light on these two mechanisms as part of multimodal communication. Specifically, we examine gaze behaviour (direction, duration up to the production of vocal feedback, and possible shift afterwards) in both the feedback producer (the listener) and feedback receiver (the speaker) when two types of turn-regulating vocal feedback are produced. These two types of feedback refer to feedback when the feedback producer does or does not subsequently take the floor. We additionally compare results across two visibility conditions: when the visual channel is available (with eye contact) and when it is not available (without eye contact). The aim of the latter manipulation is to investigate how far the overall gaze and feedback behaviour, as well as the relationship between them, changes when only the auditory channel is available.</p>
<p>Below, we review research on eye gaze and turn alternation in Section (2.1) and vocal feedback in Section (2.2) and discuss the interplay of vocal and visual cues for turn alternation in Section (2.3). Since we are concerned with the difference between dialogue with and without eye contact, we review research on the impact of visibility of the interlocutor in Section (2.4). In Section (3), we list and discuss our research questions; in Section (4), we provide information on the materials and methods; in Section (5), we present the results; and finally, in Sections (6) and (7), we discuss our findings in relation to previous findings, as well as limitations and future directions, respectively.</p>
</sec>
<sec id="sec2">
<label>2</label>
<title>Background</title>
<sec id="sec3">
<label>2.1</label>
<title>Eye gaze and turn alternation</title>
<p>When individuals engage in social interactions, these exchanges are naturally organised into distinct segments known as turns, representing the time during which one person speaks before the other takes the floor. Early studies have looked at conversational partners&#x2019; gaze in relation to their alternating roles as speaker and listener (as in <xref ref-type="bibr" rid="ref41">Goodwin, 1981</xref>, which we also refer to as primary and secondary speakers, respectively, since the listener in fact speaks when giving vocal feedback). These studies claim that in dyadic interactions, people tend to look at the other participant more when they are listening than when they are speaking (<xref ref-type="bibr" rid="ref4">Argyle and Cook, 1976</xref>, among many others). <xref ref-type="bibr" rid="ref57">Kendon (1967</xref>, <xref ref-type="bibr" rid="ref58">1990)</xref> provided a more precise description of the different patterns of speaker and listener gaze: Listeners tend to maintain long gazes at speakers, interrupted by brief glances away, while speakers alternate gazes towards and away from the listener with approximately equal gaze durations.</p>
<p>In addition to these general gaze patterns, researchers have looked more specifically at what happens at crucial conversational moments such as turn transitions. In speech, turns have generally been found to transition from one speaker to another with brief gaps and few overlaps (<xref ref-type="bibr" rid="ref63">Levinson and Torreira, 2015</xref>). This pattern, which appears to be highly robust in the face of individual, methodological, and contextual variation, raises interest in the nature of the factors that regulate this remarkable synchronisation between interlocutors.</p>
<p>The multimodal nature of turn transitions is widely attested (see, for example, <xref ref-type="bibr" rid="ref59">Kendrick et al., 2023</xref> for the effect of manual and gaze signals on turn transitions), as is the role of eye gaze in facilitating turn-taking (see <xref ref-type="bibr" rid="ref27">Degutyte and Astell, 2021</xref> for an extensive review). This was established in the seminal work of <xref ref-type="bibr" rid="ref57">Kendon (1967)</xref>, who has been highly influential in understanding the role of eye gaze at the start and end of a speaker&#x2019;s turn. In a corpus of spontaneous conversations, he found that over 70% of utterances began with the speaker looking away from the listener, while over 70% ended with the speaker looking at the listener. Kendon argued that at the beginning of their turn, participants in a conversation tend to avert their gaze (i.e., they look away from their interlocutor). This serves two functions: first, indicating that they are focusing internally, concentrating on formulating their thoughts, planning their speech, or recalling information before they begin to speak; and second, functioning as a signal to others that they intend to take the floor by initiating a speaking turn. Thus, by averting gaze, the participant who currently holds the conversational floor signals that they are not available as a <italic>listener</italic>.</p>
<p>These observations were confirmed by a number of subsequent studies on dyadic exchanges (e.g., <xref ref-type="bibr" rid="ref26">Cummins, 2012</xref>; <xref ref-type="bibr" rid="ref29">Duncan, 1972</xref>; <xref ref-type="bibr" rid="ref30">Duncan and Fiske, 1977</xref>; <xref ref-type="bibr" rid="ref47">Ho et al., 2015</xref>; <xref ref-type="bibr" rid="ref70">Oertel et al., 2012</xref>; see also <xref ref-type="bibr" rid="ref52">Jokinen et al., 2009</xref> for natural three-party conversations). For instance, <xref ref-type="bibr" rid="ref69">Novick et al. (1996)</xref> found a systematic temporal alignment between gaze and speech during turn-taking. In that study, dyads were recorded while playing guessing games using an eye tracking device. They found that at a turn transition, the speaker ends their turn looking at the listener, and the listener begins to speak with an averted gaze. Although Kendon&#x2019;s observations have been confirmed by the aforementioned studies, a number of studies have attributed gaze behaviour during turn-taking to different factors (<xref ref-type="bibr" rid="ref13">Beattie, 1978</xref>; <xref ref-type="bibr" rid="ref77">Rossano, 2012</xref>; <xref ref-type="bibr" rid="ref80">Rutter et al., 1978</xref>; <xref ref-type="bibr" rid="ref94">Streeck, 2014</xref>). <xref ref-type="bibr" rid="ref13">Beattie (1978)</xref> argued that the speaker&#x2019;s gaze away during early utterance production (and during re-engagement at the end of the utterance) is solely driven by the necessity to reduce cognitive load and does not have any regulatory role during turn-taking. Alternatively, <xref ref-type="bibr" rid="ref78">Rossano et al. (2009)</xref> and <xref ref-type="bibr" rid="ref94">Streeck (2014)</xref> claimed that gaze does not facilitate turn-taking as such, but it does facilitate the organisation of complex actions that might require multiple turns to complete.</p>
<p>At the end of a turn, <xref ref-type="bibr" rid="ref57">Kendon (1967)</xref> proposed that participants gaze towards the listener with the function of checking their availability as the next speaker and, thus, as a signal of turn-yielding. This proposal is strongly supported in later studies (<xref ref-type="bibr" rid="ref69">Novick et al., 1996</xref>; <xref ref-type="bibr" rid="ref80">Rutter et al., 1978</xref>; <xref ref-type="bibr" rid="ref62">Lerner, 2003</xref>; <xref ref-type="bibr" rid="ref52">Jokinen et al., 2009</xref>, <xref ref-type="bibr" rid="ref51">2013</xref>; <xref ref-type="bibr" rid="ref47">Ho et al., 2015</xref>; <xref ref-type="bibr" rid="ref20">Br&#x00F4;ne et al., 2017</xref>; <xref ref-type="bibr" rid="ref8">Auer, 2018</xref>; <xref ref-type="bibr" rid="ref17">Blythe et al., 2018</xref>; <xref ref-type="bibr" rid="ref94">Streeck, 2014</xref>; <xref ref-type="bibr" rid="ref70">Oertel et al., 2012</xref>; <xref ref-type="bibr" rid="ref56">Kawahara et al., 2012</xref>; for group conversations see also <xref ref-type="bibr" rid="ref42">Harrigan and Steffen, 1983</xref>; <xref ref-type="bibr" rid="ref54">Kalma, 1992</xref>). However, Kendon&#x2019;s findings were not supported across the board in this context either. <xref ref-type="bibr" rid="ref80">Rutter et al. (1978)</xref> commented that Kendon&#x2019;s predictions about floor changes can only occur if the listener is also looking at the speaker, that is, in the case of mutual gaze between the two participants. Moreover, <xref ref-type="bibr" rid="ref69">Novick et al. (1996)</xref> observed two prevalent patterns of gaze direction during turn-taking, categorised as &#x2018;mutual-break&#x2019; and &#x2018;mutual-hold&#x2019;. At the end of an utterance, the primary speaker (the participant currently holding the floor) looks at the secondary speaker (the participant currently listening), at which point the gaze is momentarily mutual until the secondary speaker breaks the mutual gaze and starts to speak, hence &#x2018;mutual-break&#x2019;, which is the most frequent pattern. Alternatively, in instances of &#x2018;mutual-hold&#x2019;, mutual gaze is held while the secondary speaker starts speaking without immediately averting their gaze. <xref ref-type="bibr" rid="ref13">Beattie (1978)</xref>, in fact, found the converse to <xref ref-type="bibr" rid="ref57">Kendon (1967)</xref>, with more immediate speaker switches when utterances terminated with no gaze than with gaze, and no differences in the effect of gaze on the number of immediate speaker switches for complete utterances, as might be expected if gaze served a floor-allocation function.</p>
<p><xref ref-type="bibr" rid="ref27">Degutyte and Astell (2021)</xref> argued that different results relative to turn boundaries and gaze might be due to different experimental designs and different types of interaction (e.g., dyadic vs. triadic/multiparty interaction, question/answer sequences vs. other types of adjacency pairs in free conversation) and methodology (see also considerations of the same kind in <xref ref-type="bibr" rid="ref92">Spaniol et al., 2023</xref>). In addition, many individual factors can affect gaze behaviour during conversation (gender in <xref ref-type="bibr" rid="ref5">Argyle and Dean, 1965</xref>; <xref ref-type="bibr" rid="ref66">Myszka, 1975</xref>; <xref ref-type="bibr" rid="ref16">Bissonnette, 1993</xref>; social status in <xref ref-type="bibr" rid="ref66">Myszka, 1975</xref>; <xref ref-type="bibr" rid="ref36">Foulsham et al., 2010</xref>; acquaintance status in <xref ref-type="bibr" rid="ref80">Rutter et al., 1978</xref>; <xref ref-type="bibr" rid="ref16">Bissonnette, 1993</xref>; and cultural background in <xref ref-type="bibr" rid="ref78">Rossano et al., 2009</xref>).</p>
<p>While there is extensive research on eye gaze during turn alternations, there is less research on how eye gaze interacts with other aspects of turn alternation. One such aspect is turn-regulating vocal feedback, which is discussed in the next section.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Vocal feedback and turn-regulating functions</title>
<p>The speaker&#x2019;s planning of their contribution to a conversation is significantly influenced by a crucial variable: the listener&#x2019;s reaction to their utterance, realised through vocal feedback signals. While often overlooked in conversation (<xref ref-type="bibr" rid="ref90">Shelley and Gonzalez, 2013</xref>), these vocal feedback signals play a pivotal role in guiding verbal exchange, indicating the listener&#x2019;s attitude towards what is being said and conveying intentions related to floor management. Commonly referred to as &#x2018;backchannels&#x2019; or &#x2018;response tokens&#x2019;, these signals are argued to enhance fluency in social interactions by supporting the ongoing turn of the interlocutor (<xref ref-type="bibr" rid="ref2">Amador-Moreno et al., 2013</xref>) and structuring dyadic conversations (<xref ref-type="bibr" rid="ref60">Kraut et al., 1982</xref>; <xref ref-type="bibr" rid="ref81">Sacks et al., 1974</xref>; <xref ref-type="bibr" rid="ref89">Schegloff, 1982</xref>).</p>
<p>There is no consensus in the literature about their definition (<xref ref-type="bibr" rid="ref79">R&#x00FC;hlemann, 2007</xref>). <xref ref-type="bibr" rid="ref37">Fries (1952)</xref> is credited as one of the earliest to identify these &#x2018;signals of attention&#x2019; which do not disrupt the speaker&#x2019;s discourse in telephone conversations. Over time, various terms have been employed to characterise this phenomenon, including &#x2018;accompaniment signals&#x2019; (<xref ref-type="bibr" rid="ref57">Kendon, 1967</xref>), &#x2018;receipt tokens&#x2019; (<xref ref-type="bibr" rid="ref45">Heritage, 1984</xref>), &#x2018;minimal responses&#x2019; (<xref ref-type="bibr" rid="ref34">Fellegy, 1995</xref>), &#x2018;reactive tokens&#x2019; (<xref ref-type="bibr" rid="ref23">Clancy et al., 1996</xref>), &#x2018;response tokens&#x2019; (<xref ref-type="bibr" rid="ref38">Gardner, 2001</xref>), &#x2018;engaged listenership&#x2019; (<xref ref-type="bibr" rid="ref61">Lambertz, 2011</xref>), and &#x2018;active listening responses&#x2019; (<xref ref-type="bibr" rid="ref91">Simon, 2018</xref>). <xref ref-type="bibr" rid="ref99">Yngve (1970)</xref> introduced the term &#x2018;backchannel communication&#x2019; to distinguish the primary channel used by the speaker holding the floor from the one used by the listener to convey essential information without actively taking a turn. Essentially, backchannels represent tokens used to express acknowledgement and understanding, encouraging the main speaker to continue (e.g., <xref ref-type="bibr" rid="ref43">Hasegawa, 2014</xref>, among others).</p>
<p>Later, <xref ref-type="bibr" rid="ref0040">Jefferson (1983)</xref> claimed that the use of such &#x2018;acknowledgement tokens&#x2019; does not exclusively signal turn-yielding but can also cue the intention of taking the floor after acknowledging the current speaker&#x2019;s turn. In particular, she observed that some token types &#x201C;exhibit a preparedness to shift from recipiency to speakership&#x201D; (1983:4), whereas others are more systematically used as turn-yielding signals. For English, she noticed that speakers tend to use tokens such as &#x2018;mh-mh&#x2019; to signal the intention of acknowledging without taking the floor, whereas tokens such as &#x2018;yeah&#x2019; when a recipient is moving into speakership. In line with this distinction, some later studies (<xref ref-type="bibr" rid="ref28">Drummond and Hopper, 1993</xref>; <xref ref-type="bibr" rid="ref53">Jurafsky et al., 1998</xref>; <xref ref-type="bibr" rid="ref82">Savino, 2010</xref>, <xref ref-type="bibr" rid="ref83">2011</xref>, <xref ref-type="bibr" rid="ref84">2012</xref>; <xref ref-type="bibr" rid="ref86">Savino and Refice, 2013</xref>; <xref ref-type="bibr" rid="ref97">Wehrle, 2023</xref>; <xref ref-type="bibr" rid="ref50">Janz, 2022</xref>; <xref ref-type="bibr" rid="ref87">Sbranna et al., 2022</xref>; <xref ref-type="bibr" rid="ref92">Spaniol et al., 2023</xref>; <xref ref-type="bibr" rid="ref88">Sbranna et al., 2024</xref>) categorised backchannels based on two turn-taking functions, that is, passive recipiency (henceforth PR) and incipient speakership (henceforth IS). Passive recipiency involves tokens produced without the speaker taking the floor, serving as acknowledgements and continuers (i.e., backchannels, as intended by <xref ref-type="bibr" rid="ref99">Yngve, 1970</xref>). Incipient speakership refers to tokens used by a speaker to acknowledge the interlocutor&#x2019;s turn before producing a new turn themselves, realising a turn transition. This classification has also been partially supported by intonation research, which has observed a tendency for certain tokens and intonation contours to co-occur with these functions (<xref ref-type="bibr" rid="ref82">Savino, 2010</xref>, <xref ref-type="bibr" rid="ref83">2011</xref>; <xref ref-type="bibr" rid="ref87">Sbranna et al., 2022</xref>; <xref ref-type="bibr" rid="ref88">Sbranna et al., 2024</xref>).</p>
<p>In this study, we adopt this operationalisation of the feedback turn-taking function, as it accounts for the role that vocal feedback signals play in the complex and multidimensional turn management system, alongside feedback signals of a different (i.e., visual) nature.</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>The interplay of vocal and visual cues for turn alternation</title>
<p>Given that feedback is meant to support the ongoing turn of the primary speaker, this speaker can actually invite the secondary speaker to produce feedback. This invitation is multimodal in nature and can be signalled in different ways, for example, by inserting a pause by intonation, or head or gaze movements. <xref ref-type="bibr" rid="ref44">Heldner et al. (2013)</xref> proposed that the primary speaker provides &#x201C;backchannel relevance spaces&#x201D; to allow the listener to insert backchannels, although in their corpus of face-to-face interaction, the number of such spaces was considerably greater than the actual backchannels provided. The authors argue that not all backchannel relevance spaces are filled (either with vocal or visual feedback) as listeners choose feedback positions that actively support speakers in constructing their discourse.</p>
<p>Some studies on the interaction between vocal feedback and, in particular, eye gaze have shown that vocal and non-vocal feedback signals are used to show acknowledgement and understanding of the primary speaker without taking the floor&#x2014;those which we refer to as PR tokens, and Heldner&#x2019;s &#x2018;backchannels&#x2019; (2013)&#x2014;occur during mutual gaze between a listener and a speaker, in accordance with a number of other studies (<xref ref-type="bibr" rid="ref57">Kendon, 1967</xref>; <xref ref-type="bibr" rid="ref10">Bavelas et al., 2002</xref>; <xref ref-type="bibr" rid="ref31">Eberhard and Nicholson, 2010</xref>; <xref ref-type="bibr" rid="ref26">Cummins, 2012</xref>; <xref ref-type="bibr" rid="ref70">Oertel et al., 2012</xref>). For example, <xref ref-type="bibr" rid="ref70">Oertel et al. (2012)</xref> compared gaze behaviour in dyadic interactions at turn transitions during speech overlap, in silence, and in the vicinity of backchannels and found that the production of vocal feedback signals is associated with an increase in mutual gaze. At a more detailed level, <xref ref-type="bibr" rid="ref10">Bavelas et al. (2002)</xref> found that gaze patterns used to coordinate feedback signals in dialogue are often initiated by the speaker gazing at the listener, who then looks back, resulting in short periods of mutual gaze broken by the listener looking away shortly afterwards. The authors precisely examined when and how listeners insert their feedback into a speaker&#x2019;s narrative by involving participants in a storytelling task. A collaborative approach would predict a relationship between the speaker&#x2019;s acts and the listener&#x2019;s responses, and the authors proposed that gaze coordinates this collaboration. They found that the listener typically looks at the speaker more often than the other way around. However, at key points in their speech, the speaker seeks a response by looking at the listener, creating a brief period of mutual gaze, called &#x201C;gaze window&#x201D; by the authors. They observed that, during gaze windows, the listener was very likely to provide a reaction, such as a vocal &#x2018;mhm&#x2019; or a nod, after which the speaker interrupted the gaze window by quickly looking away and continuing to speak. This pattern aligns with previous studies on the use of multimodal cues in the perception of interrogativity, which show that gaze direction towards the interlocutor and other gestural signals enhance the perception of a response being expected (<xref ref-type="bibr" rid="ref18">Borr&#x00E0;s-Comes et al., 2014</xref>).</p>
<p>Although mutual gaze has been found to be a strong predictor of feedback signals independent of their modality (<xref ref-type="bibr" rid="ref35">Ferr&#x00E9; and Renaudier, 2017</xref>; <xref ref-type="bibr" rid="ref46">Hjalmarsson and Oertel, 2012</xref>; <xref ref-type="bibr" rid="ref73">Poppe et al., 2011</xref>), a difference has been reported between vocal and non-vocal (i.e., head gestures) feedback signals, with non-vocal signals being more often produced in correspondence with mutual gaze than vocal signals (<xref ref-type="bibr" rid="ref35">Ferr&#x00E9; and Renaudier, 2017</xref>; <xref ref-type="bibr" rid="ref31">Eberhard and Nicholson, 2010</xref>; <xref ref-type="bibr" rid="ref96">Truong et al., 2011</xref>). This is probably because vocal signals do not need to be conveyed through the visual channel. Moreover, gaze was found to be sustained more often throughout a sequence that contained a visual backchannel than in a sequence containing a verbal backchannel (<xref ref-type="bibr" rid="ref35">Ferr&#x00E9; and Renaudier, 2017</xref>). The strong correlation between being looked at and gestural backchannels, but not vocal ones (<xref ref-type="bibr" rid="ref15">Bertrand et al., 2007</xref>), was motivated by the idea that gaze might establish a communication mode between interlocutors. Nonetheless, in contexts where the primary speaker was not gazing at the interlocutor, both gestural and/or vocal backchannels were produced.</p>
<p><xref ref-type="bibr" rid="ref71">Ond&#x00E1;&#x0161; et al. (2023)</xref> also looked at the temporal detail of the interval between the start of the gaze directed towards the interlocutor and the subsequent backchannel in an interview scenario and found that, in most cases, the moderator received vocal feedback from his guest within 500&#x202F;ms of initialising direct eye contact. The second most probable time interval was 1,500&#x2013;2000&#x202F;ms. This same interval was found to be the most probable in the opposite scenario, that is, the moderator providing feedback to the guest. One possible speculation on this different behaviour is that the moderator is meant to lead the conversation, thus getting quicker reactions from the guest, who is expected to inform the moderator that they acknowledge and follow the dialogue structure. However, this result is linked to a very specific conversational format, which probably has its own dynamics that are different from other formats of conversation. Moreover, <xref ref-type="bibr" rid="ref59">Kendrick et al. (2023)</xref>, analysing the impact of visual cues in turn-timing, found that turn transitions were sped up by the use of manual gestures but not by gaze behaviour.</p>
<p>These studies focused on the interaction between eye gaze and backchannels, accounting only for their PR function. To our knowledge, only one recent exploratory study (<xref ref-type="bibr" rid="ref92">Spaniol et al., 2023</xref>) has investigated the interplay of gaze patterns and turn-regulating vocal feedback using the same PR vs. IS paradigm for vocal feedback. In particular, the authors analysed gaze and vocal feedback behaviour in dyadic conversations across different communicative contexts, namely, (1) a free conversation in which participants get to know each other, (2) a Tangram game-based interaction, and (3) a free discussion about the game participants had just played together. In contrast to previous findings, they found that in the task-based context, vocal feedback was mostly provided in conjunction with averted gaze, independently of its turn-regulating function. They explained their results by the presence of a visual competitor during the conversation, that is, the task material. In the other two contexts of free conversation, three out of four dyads produced PR feedback signals more under gaze directed to the interlocutor than IS feedback signals, showing that the turn-regulating function of backchannels distributes differently on gaze behaviour patterns and that this distribution is influenced by the context of the conversation.</p>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>The impact of (non-)visibility on communicative channels</title>
<p>Another important question for the present study concerns the use of speech and gestures when the interlocutor is not visible.</p>
<p>Some studies have investigated the use of manual gestures in conversations, with and without visibility between participants, concluding that gestures have both communicative and internal cognitive functions. For example, <xref ref-type="bibr" rid="ref1">Alibali et al. (2001)</xref> found that participants were still using representational gestures, that is, gestures depicting semantic content related to speech, when an opaque panel hindered mutual visibility, although to a lesser extent than in visibility conditions. Interestingly, the number of beat gestures, that is, simple rhythmic gestures lacking semantic content, remained consistent across both conditions. <xref ref-type="bibr" rid="ref12">Bavelas et al. (2008)</xref> compared the use of gesture in face-to-face and telephone conversations and concluded that, instead of being incidental, the gestures made while talking on the phone appeared to be a deliberate adjustment to a dialogue without visual contact. Indeed, both the visibility and non-visibility groups gestured at a high rate, although in the visibility condition, participants gestured with forms and relationships with words that were more informative for their listeners. These studies point to similar conclusions: first, that gestures are adjusted to the interlocutor&#x2019;s need; second, that gestures fulfil not only a communicative function but also an internal cognitive function (see also <xref ref-type="bibr" rid="ref65">McNeill, 2005</xref>).</p>
<p>Based on these findings, an interesting question is whether these results apply to eye gaze. <xref ref-type="bibr" rid="ref6">Argyle et al. (1973)</xref> sought to distinguish the various functions of gaze by utilising a one-way screen to manipulate the conditions of speakers observing or being observed. They found evidence of gaze having monitoring and signalling functions. In the first case, they observed that individuals who could see through a one-way screen exhibited increased gaze while speaking compared to those without visual access; in the second case, they observed that even the individuals who could not see their interlocutor still directed their gaze towards the other person occasionally while speaking. Indeed, several studies have found that, even in the absence of visibility, listeners are particularly sensitive to sound sources directed straight at them and are able to recognise the speaker&#x2019;s head orientation, that is, the situation corresponding to mutual gaze in visible conditions (<xref ref-type="bibr" rid="ref32">Edlund et al., 2012</xref>; <xref ref-type="bibr" rid="ref55">Kato et al., 2010</xref>; <xref ref-type="bibr" rid="ref67">Nakano et al., 2008</xref>).</p>
<p>In a study examining mutual gaze patterns in dyadic conversation under light and darkness conditions, <xref ref-type="bibr" rid="ref75">Renklint et al. (2012)</xref> found a notable reduction in instances of mutual gaze in the absence of light. To interpret this result, the authors claimed that mutual gaze is &#x201C;made for the other person to see&#x201D; (aligning with the perspective of <xref ref-type="bibr" rid="ref9">Bavelas et al., 1992</xref>:483) and that gaze extends beyond mere learned behaviour, having its own functions and meanings. Moreover, the authors noticed that in dyadic conversations, the addressee is consistently the only other participant in the conversation, and given an implicit mutual agreement about an alternation of speakers in a conversation, selecting the next speaker becomes a mere formality. This is very different from multiparty conversations, in which gaze plays a significant role in turn alternation. Thus, they suggested that in dyadic conversations, information derived from linguistic and phonetic aspects (e.g., sentence structure and prosody) provides enough cues to predict turn alternation, making mutual gaze less essential for successful communication. From these studies, it appears that (i) gaze clearly has a dual function: sensing (i.e., sampling information from a visual scene) and signalling (i.e., not just to gain information from the environment, but with an explicit and deliberate communicative goal) (<xref ref-type="bibr" rid="ref4">Argyle and Cook, 1976</xref>; <xref ref-type="bibr" rid="ref39">Gobel et al., 2015</xref>; <xref ref-type="bibr" rid="ref76">Risko et al., 2016</xref>; <xref ref-type="bibr" rid="ref22">Ca&#x00F1;igueral and Hamilton, 2019</xref>), and (ii) in the absence of visibility, gazing at the position where the conversational partner is known to be is reduced but is not absent. This might be due to the fact that in such a condition only, the sensing function of gaze applies.</p>
<p>The effect of non-visibility also relates to the fluency of the speakers&#x2019; individual speech and the turn alternation system, both of which vocal feedback crucially contributes to. <xref ref-type="bibr" rid="ref14">Beattie (1979)</xref> reported that the absence of gaze leads to an increased use of filled pauses (see also <xref ref-type="bibr" rid="ref25">Cook and Lalljee, 1972</xref>) and interruptions (<xref ref-type="bibr" rid="ref7">Argyle et al., 1968</xref>). Similar results were obtained by <xref ref-type="bibr" rid="ref19">Boyle et al. (1994)</xref>, who studied the effect of visibility on task resolution. The authors found that in the visibility condition, there is greater efficiency in the dialogues attributed to the exchange of visually transmitted non-verbal signals, while in the non-visibility condition, oral communication is characterised as having greater flexibility and versatility. In the latter case, participants tried to compensate for the absence of the visual channel by interrupting their partners more frequently and using more vocal backchannels to support the primary speaker. In addition, <xref ref-type="bibr" rid="ref68">Neiberg and Gustafson (2011)</xref>, comparing eye-contact to non-eye-contact conversations, found an increase in overlapped turn transitions in the absence of visibility, which they explain by a possible increase in unintentional interruptions. From these findings, it seems that visibility improves the smoothness of the dialogue flow, but, in turn, in the absence of visibility, speakers rely more on vocal resources, especially feedback signals.</p>
</sec>
</sec>
<sec id="sec7">
<label>3</label>
<title>Research questions</title>
<p>Previous research has established the importance of eye gaze, particularly eye gaze direction, for control and coordination in human interactions. However, little is known about the interplay between eye gaze and vocal feedback encouraging the interlocutor to continue speaking (Passive Recipiency), and feedback initiating a turn (Incipient Speakership). We investigate this relationship in dialogues between Italian speakers playing a Tangram game. Importantly, this game was played in two visibility conditions, with and without eye contact, to compare how turn-taking is regulated with and without gaze being used for this purpose.</p>
<p>In particular, we aim to answer the following research questions:</p>
<list list-type="order">
<list-item>
<p>How do gaze and vocal turn-regulating feedback work together?</p>
</list-item>
<list-item>
<p>What is the effect of (non-)visibility of the interlocutor on the feedback-gaze relation?</p>
</list-item>
</list>
<p>Our first research question is based on the findings reported above that turn initiation preferably co-occurs with speakers tending to gaze away from their interlocutor and turn yielding with speakers looking at their interlocutor. Given that IS vocal feedback initiates a turn, we predict behaviour similar to that reported for turn initiation. By analogy, since PR vocal feedback producers do not take the turn from the interlocutor, we expect similar behaviour to that found for turn yielding. We provide our predictions for the feedback producer and receiver separately as follows:</p>
<list list-type="simple">
<list-item>
<p>1a&#x00A0;&#x00A0;Feedback producer (listener or secondary speaker):</p></list-item>
</list>
<list list-type="bullet">
<list-item>
<p>IS vocal feedback, being turn-initiating, is produced while looking away from the feedback receiver to signal unavailability as a listener and commitment to speech planning.</p>
</list-item>
<list-item>
<p>PR vocal feedback, which does not involve turn transition, is produced while looking at the primary speaker to show availability as a listener.</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>1b&#x00A0;&#x00A0;Feedback receiver (primary speaker):</p></list-item>
</list>
<list list-type="bullet">
<list-item>
<p>If gaze is primarily used to request feedback, vocal feedback should be produced when/just after the primary speaker looks at the secondary speaker.</p>
</list-item>
<list-item>
<p>If gaze at the conversational partner is intended to signal turn yielding, the feedback receiver is expected to look away when receiving PR feedback so as not to yield the turn, but to look at the partner when receiving IS feedback, to signal their availability as a listener and thus their willingness to yield the turn.</p>
</list-item>
</list>
<p>In the case of the feedback receiver, it is important to keep in mind that the signalling function can be two-fold: feedback request and/or turn holding/yielding, and that the two communicative intentions can overlap, resulting in a more complex picture. Although the feedback receiver is not in control of the actions of the feedback producer, such that speakers may not align in the signals they send with regard to willingness to yield and take the turn, we predict that their behaviour will be somehow cooperative. Based on similar measures previously used in the literature, we explore this cooperativeness through:</p>
<list list-type="simple">
<list-item>
<p>1c&#x00A0;&#x00A0;The temporal details of gaze direction intervals (i.e., intervals of time in which participants hold their gaze in a specific direction, delimited by gaze shifts from and to another direction. See <xref ref-type="fig" rid="fig1">Figure 1</xref> in the Method section) that induce vocal feedback to assess whether gaze directed to the addressee speeds up feedback production and/or turn alternation, a topic about which the literature does not provide enough information to build expectations (see Section 2.3).</p>
</list-item>
<list-item>
<p>1a&#x00A0;&#x00A0;The shift in gaze direction after vocal feedback utterance. We expect that the gaze window will close after vocal feedback, as suggested by previous studies. In other words, if gaze towards the addressee is present before the vocal feedback signal, it shifts to the task afterwards.</p>
</list-item>
</list>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Example of all annotation tiers in ELAN. The labels for Role are either &#x201C;D&#x201D; for Director or &#x201C;M&#x201D; for Matcher. The labels for Gaze are &#x201C;T&#x201D; for task-directed gaze, &#x201C;A&#x201D; for addressee-oriented gaze, or &#x201C;O&#x201D; for other, under which we define gaze to the experimenter. In the tier &#x201C;ACK&#x201D; (acknowledgements), the turn-regulating function of the vocal feedback is annotated as IS for Incipient Speakership and PR for Passive Recipiency.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g001.tif"/>
</fig>
<p>Relative to our second research question, based on previous findings, we expect that in the non-visibility condition:</p>
<list list-type="simple">
<list-item>
<p>2a&#x00A0;&#x00A0;We will not find many co-occurrences of gaze and vocal feedback as multimodal cues for turn regulation, with only the vocal channel being available.</p>
</list-item>
<list-item>
<p>2b&#x00A0;&#x00A0;Gaze towards the addressee will be drastically reduced.</p>
</list-item>
<list-item>
<p>2c&#x00A0;&#x00A0;Conversely, the use of vocal feedback will be enhanced to compensate for the lack of visual cues.</p>
</list-item>
</list>
<p>These expectations are linked to the assumption that gaze has a dual function of sensing and signalling and that the latter is only possible when mutual gaze is available.<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref></p>
</sec>
<sec sec-type="materials|methods" id="sec8">
<label>4</label>
<title>Methods and materials</title>
<sec id="sec9">
<label>4.1</label>
<title>Corpus</title>
<p>We analysed interactions from an Italian corpus of dyadic task-oriented conversations (described in <xref ref-type="bibr" rid="ref85">Savino et al., 2018</xref>). As this is the first multimodal analysis of the corpus, the corpus and setup are described in detail below.</p>
<sec id="sec10">
<label>4.1.1</label>
<title>Participants</title>
<p>The participants were 12 Italian speakers (6 dyads), all students at the University of Bari, and all from the same geo-linguistic area (the Bari district in Apulia, a southeastern region of Italy). They were all young female adults (aged 21&#x2013;25&#x202F;years) and university classmates, ensuring a degree of familiarity within the dyads. Keeping these factors constant was crucial since gender and acquaintance status have been shown to affect gaze behaviour (see <xref ref-type="bibr" rid="ref66">Myszka, 1975</xref> and <xref ref-type="bibr" rid="ref16">Bissonnette, 1993</xref>, respectively). All speakers voluntarily participated in the experiment and signed an informed consent form. They obtained a course credit for participating in the experiment.</p>
</sec>
<sec id="sec11">
<label>4.1.2</label>
<title>Elicitation method</title>
<p>Participants were asked to play a tangram-based matching game, organised in 22 rounds. For each game round, the two participants were given sets of Tangram figures according to their role in a particular game round: Director or Matcher.<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> The Director was provided with a sheet with four Tangram figures on it, one of which was marked by an arrow, and the Matcher was given another sheet with only one of the figures belonging to the Director&#x2019;s set (an example of both types of figure is provided in <xref ref-type="fig" rid="fig2">Figure 2</xref>). Participants were unable to see their partner&#x2019;s figure(s), and the goal of each game round was to establish whether the Tangram figure given to the Matcher corresponded to the figure marked by the arrow in the Director&#x2019;s set. A round is defined as each game dialogue segment starting from when participants uncover a set of Tangram figures and finishing when they reach their joint decision as to the matching/mismatching for that set.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Example of a Tangram set of figures used by players in a game round. The picture on the left is the set of four figures for the Director, whereby one figure is indicated by an arrow. The picture on the right is the figure provided to the Matcher. The goal of a game round is to decide together whether the figure on the right corresponds to the figure on the left indicated by an arrow.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g002.tif"/>
</fig>
<p>In the instructions, the Director was first asked to describe the figure indicated by an arrow, after which the participants could exchange information about the shape of their respective figures. Information exchange could be freely and spontaneously managed by the participants until an agreement was reached regarding whether the two figures were the same. A typical game round started with the Director describing their Tangram figure (indicated with an arrow) to the Matcher and ends with the two participants checking whether their joint decision about the matching/unmatching figures was correct. The players were explicitly instructed to arrive at a decision based on common agreement. To encourage cooperative behaviour, they were told that they would both score a point every time they correctly identified whether the figure was the same or not and that they would both lose a point when their guess was wrong.</p>
</sec>
<sec id="sec12">
<label>4.1.3</label>
<title>Recording sessions</title>
<p>During the recordings, the two participants sat at separate desks facing each other. A Panasonic HC-V700 camcorder was placed behind each participant for the video recording. The camcorder was placed to clearly capture the participant&#x2019;s visible upper body and face (see <xref ref-type="fig" rid="fig3">Figure 3</xref>).</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Example of the recording setup from the perspective of the two cameras in the Eye-Contact condition.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g003.tif"/>
</fig>
<p>In the eye-contact (henceforth EC) condition, in which the two participants <italic>were able to see</italic> each other, a low opaque panel was placed between the participants&#x2019; desks at a suitable height to prevent them from seeing each other&#x2019;s Tangram figures while preserving eye contact between players. In the no-eye-contact (henceforth NEC) condition, in which the two participants <italic>were not able to see</italic> each other, a higher opaque panel was placed between the participants&#x2019; desks so that the conversational partner was completely hidden. The NEC game sessions were recorded 1&#x202F;month after the EC sessions. The sets of Tangram figures differed across the two sessions.</p>
<p>A total of 12 recordings were obtained (six dyads &#x00D7; two conditions): two recordings for each dyad, one in the EC and one in the NEC condition. Each recording lasted approximately 30&#x202F;min. All recordings were carried out in a quiet room at the Department of Education, Psychology, and Communication of Bari University, Italy.</p>
<p>In the present study, we analyse a subset of this corpus, consisting of six rounds per dyad in each of the two visibility conditions.</p>
</sec>
<sec id="sec13">
<label>4.1.4</label>
<title>Annotation of vocal feedback expressions</title>
<p>Manual annotation of the speech signal was carried out at various levels in Praat (<xref ref-type="bibr" rid="ref0020">Boersma and Weenink 2001</xref>), including intervals corresponding to game rounds, inter-pausal units (IPUs, defined as portions of speech delimited by at least 100&#x202F;ms of silence), phonological words, and syllables.</p>
<p>A specific annotation tier (see <xref ref-type="fig" rid="fig1">Figure 1</xref>) is devoted to the description of vocal feedback expressions in relation to their turn-regulating function. In this tier, all lexical tokens, such as <italic>s&#x00EC;</italic> (yes), <italic>esatto</italic> (exactly) <italic>va bene</italic> (alright), and <italic>okay</italic>, and non-lexical tokens, such as <italic>mh-mh</italic> and <italic>eh</italic>, signalling attention/understanding/acknowledgement of the current speaker were annotated as feedback signals. To code their turn-regulating functions, we adopted the pragmatic distinction between PR and IS described in the literature review (Section 2.2), resulting in the following operational criteria:</p>
<list list-type="order">
<list-item>
<p>When the secondary speaker did not take the floor after producing the vocal feedback (so that the primary speaker continued talking), that is, the vocal feedback signal was <italic>not followed</italic> by a turn alternation, the feedback expression was coded as fulfilling a PR function.</p>
</list-item>
<list-item>
<p>When the secondary speaker took the floor after producing the vocal feedback, that is, the feedback production was <italic>immediately followed</italic> by a turn alternation, that feedback expression was coded as fulfilling an IS function.</p>
</list-item>
</list>
<p>These labelling criteria are exemplified in the following example extracted from the corpus:</p>
<disp-quote>
<p>Director: &#x003C;ehm&#x003E;&#x202F;sulla destra c&#x2019;&#x00E8;&#x202F;&#x003C;ee&#x003E;&#x202F;un triangolo rettangolo.</p>
<p>
<italic>(Eng.: &#x003C;um&#x003E; on the right there is &#x003C;er&#x003E; a right-angled triangle)</italic>
</p>
<p>Matcher: &#x003C;m&#x003E;&#x202F;[vocal feedback-PR].</p>
<p>Director: cio&#x00E8; su&#x202F;&#x003C;ehm&#x003E;&#x202F;mentre la base &#x00E8; formata da un parallelepipedo a sinistra e un triangolo con la punta rivolta verso il basso.</p>
<p>
<italic>(Eng.: that is on &#x003C;uhm&#x003E; while the base is formed by a parallelepiped to the left and a triangle with the tip of the triangle facing downwards)</italic>
</p>
<p>Matcher: okay [vocal feedback-IS] la base ci siamo descrivimi la vela.</p>
<p>
<italic>(Eng.: okay for the base we are there, describe the sail)</italic>
</p>
</disp-quote>
</sec>
<sec id="sec14">
<label>4.1.5</label>
<title>Annotation of eye gaze direction intervals</title>
<p>The annotation of gaze was performed using ELAN (<xref ref-type="bibr" rid="ref95">The Language Archive, 2023</xref>; <xref ref-type="bibr" rid="ref98">Wittenburg et al., 2006</xref>). Following the methodology adopted in previous studies (<xref ref-type="bibr" rid="ref57">Kendon, 1967</xref>; <xref ref-type="bibr" rid="ref13">Beattie, 1978</xref>; <xref ref-type="bibr" rid="ref40">Goodwin, 1980</xref>; <xref ref-type="bibr" rid="ref33">Egbert, 1996</xref>; <xref ref-type="bibr" rid="ref69">Novick et al., 1996</xref>; <xref ref-type="bibr" rid="ref52">Jokinen et al., 2009</xref>; <xref ref-type="bibr" rid="ref94">Streeck, 2014</xref>; <xref ref-type="bibr" rid="ref8">Auer, 2018</xref>; <xref ref-type="bibr" rid="ref17">Blythe et al., 2018</xref>), we annotated intervals of specific gaze direction, namely, the time intervals in which participants gaze continuously in a specific direction without a perceivable change, and where gaze boundaries are identified by a gaze shift from and to a different direction. Gaze direction intervals were equally defined in the EC and NEC conditions for ease of comparison: &#x201C;Task,&#x201D; &#x201C;Addressee,&#x201D; or &#x201C;Other.&#x201D;</p>
<p>In the EC condition, gaze to &#x201C;Task&#x201D; was annotated when the participant gazed at the table, where their own sheet with the Tangram figures was positioned. Gaze to &#x201C;Addressee&#x201D; was annotated when the participant was looking straight ahead at the other participant. Finally, gaze to &#x201C;Other&#x201D; includes gaze directed towards the experimenter.</p>
<p>In the NEC condition, gaze to &#x201C;Task&#x201D; was annotated following the same criteria as in the eye-contact condition, whereas gaze to &#x201C;Addressee&#x201D; was interpreted as <italic>addressee</italic>-<italic>oriented</italic> gaze. To annotate it, our main criterion was a head movement upward towards the panel behind which the interlocutor was sitting, accompanied by gaze that could be directed straight ahead, to the top-right or top-left, but without a change in the head position, which remained straight, directed towards the panel. A change in pupil direction was not interpreted as attention towards something else, since it appeared that participants were focused on their interlocutor and the conversation, and that changes in pupil direction were not related to listening or speaking, as the top-right or top-left pupil movements co-occurred with both speech and silence. Our interpretation of participants&#x2019; behaviour is that they wanted to address their gaze at the interlocutor, but finding the panel in front of them made them shift their pupil direction to the top-right or left in some cases. Gaze to &#x201C;Addressee&#x201D; was also easily discernible from &#x201C;Other,&#x201D; which was again annotated when participants looked at the experimenter by clearly turning their heads to the side. Note that gaze was observed in video recordings with a camera in front of each speaker and not by means of eye-tracking glasses, which, despite providing very good visibility of the eyes, do not enable maximum precision for registering smaller details such as pupil movements. Consequently, these broadly defined categories matched our means and scope.</p>
<p>We did not include the category &#x201C;Other&#x201D; in our analysis for either of the two conditions because it is not relevant to our research question.</p>
<p>We did not establish a minimum duration threshold for eye gaze annotation; instead, we annotated all perceivable changes from one of the mentioned gaze directions to another. However, to make a comparison to studies which annotated gaze based on a minimum possible duration threshold (<xref ref-type="bibr" rid="ref14">Beattie, 1979</xref>; <xref ref-type="bibr" rid="ref52">Jokinen et al., 2009</xref>; <xref ref-type="bibr" rid="ref20">Br&#x00F4;ne et al., 2017</xref>; <xref ref-type="bibr" rid="ref100">Zima et al., 2019</xref>; <xref ref-type="bibr" rid="ref10">Bavelas et al., 2002</xref>, among others), the shortest duration value found in our corpus is 0.069&#x202F;s, which is below most thresholds previously used (see <xref ref-type="bibr" rid="ref27">Degutyte and Astell, 2021</xref> for a comprehensive list).</p>
<p>A portion of 20% of the data was annotated independently by a fellow linguist trained in eye gaze annotation on video recordings. The inter-annotator reliability of gaze direction annotations was assessed using Staccato (<xref ref-type="bibr" rid="ref64">L&#x00FC;cking et al., 2011</xref>), which is directly available in ELAN and provides a reliable measure of the temporal overlap between annotations based on Thomann&#x2019;s technique (2001). Staccato provided a mean degree of overlap of 70%, which can be interpreted as substantial reliability. To ensure a homogenous interpretation of the category labels for gaze direction, the two annotators discussed labels in the case of disagreement until a common decision was reached. The output annotations performed by the first author are retained in the analysis. An example of all annotation tiers in ELAN is shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
</sec>
</sec>
<sec id="sec15">
<label>4.2</label>
<title>Data treatment for the analysis</title>
<p>Annotation of eye gaze was performed for rounds 1, 2, 9, 10, 21, and 22, for each dyad in both conditions. The selected game rounds correspond to the beginning (rounds 1&#x2013;2), medial part (rounds 9&#x2013;19), and final part (rounds 21&#x2013;22) of the whole game session. This selection was made to control for any effect of (non-)familiarity with the task on participants&#x2019; communicative style. The total number of rounds included in the present multimodal analysis corresponds to 1.47&#x202F;h of conversation.</p>
<p>To answer our research questions (RQs), we analysed the co-occurrence of vocal feedback signals with gaze directions of both the feedback producer, or &#x201C;secondary speaker,&#x201D; and feedback receiver, or &#x201C;primary speaker&#x201D; (Section 4.1). To do so, for each participant&#x2019;s vocal feedback signal, we extracted the overlapping gaze direction intervals by both the feedback producer and feedback receiver (RQs 1.a and 1.b, Section 4.1.1) and measured the duration of these intervals up to the moment in which the vocal feedback signal is uttered (RQ 1.c, Section 4.1.2). This duration is operationalised as the time window from the onset of the overlapping gaze direction interval to the onset of the vocal feedback signal. We also extracted the gaze direction intervals overlapping the offset of each feedback signal to establish whether the production of vocal feedback would prompt a shift in gaze direction in any of the two participants (RQ 1.d, Section 4.1.3).</p>
<p>Moreover, we measured the overall amount and duration (in seconds) of gaze direction intervals (RQ 2.a, Section 4.2), as well as vocal feedback rate, operationalised as the number of feedback occurrences per minute of dialogue (RQ 2.b, Section 4.3).</p>
<p>The results of all analyses are presented across visibility conditions (RQ 2.c throughout the result sections for conciseness), i.e., eye-contact (EC) and non-eye-contact (NEC). The data that support the findings of this study, as well as the code used to perform the analysis, are openly available in the accompanying repository at <ext-link xlink:href="https://osf.io/3dqzk/" ext-link-type="uri">https://osf.io/3dqzk/</ext-link>.</p>
</sec>
</sec>
<sec sec-type="results" id="sec16">
<label>5</label>
<title>Results</title>
<p>Although a comparison between the EC and NEC conditions belongs to the second RQ, for reasons of conciseness, we present results relative to the EC and NEC conditions side by side for each measurement.</p>
<sec id="sec17">
<label>5.1</label>
<title>Gaze behaviour by the feedback producer and receiver</title>
<sec id="sec18">
<label>5.1.1</label>
<title>Occurrences of gaze direction intervals at feedback production</title>
<p><xref ref-type="fig" rid="fig4">Figure 4</xref> shows the co-occurrences of gaze direction intervals and vocal feedback productions in two graphs: the perspective of the feedback producer (&#x201C;secondary speaker,&#x201D; left panel) and the one of the feedback receiver (&#x201C;primary speaker,&#x201D; right panel).</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Occurrences of gaze direction intervals by the feedback producer and receiver at the moment in which vocal feedback signals are uttered. The proportions of gaze direction intervals are displayed as percentages on the <italic>y</italic>-axis. The two gaze direction types are colour-coded as in the legend: gaze to task (Task) in yellow and gaze towards addressee (Addressee) in green. The exact proportions and counts of occurrences are printed on the bar plots (the latter in parentheses) for the two turn-regulating functions of the vocal feedback signals, that is, Passive Recipiency (PR) and Incipient Speakership (IS), displayed on the <italic>x</italic>-axis. The results are shown for both visibility conditions, that is, eye contact (EC) and non-eye contact (NEC), as indicated on top of the bars.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g004.tif"/>
</fig>
<p>In the EC condition, the feedback producers utter most of their turn-regulating vocal feedback while looking at the task (gaze to &#x201C;Addressee&#x201D; occurs only in 12% of the cases for PR signals and 9% of the cases for IS signals), contrary to our expectations where we predicted a greater use of gaze to &#x201C;Addressee&#x201D; as a way to signal availability as listener.</p>
<p>The results in the EC condition for the feedback receiver reveal a different picture. PR signals occur in 64% of the cases when the primary speaker is looking at the task. This shows that gaze is less frequently used as a request for feedback signals (36%) and confirms our expectations that PR feedback signals are mainly uttered when the primary speaker is not looking at the secondary speaker, not signalling availability for ceding the floor. IS signals occur in equal proportions for both gaze directions: 49% of the cases in which the primary speaker is looking at the task and 51% of the cases in which the primary speaker is looking at the secondary speaker, whereby we expected the latter case to be prevalent if gaze is mainly used to signal turn management. However, we can still observe that being looked at prompts more IS than PR feedback signals, which is in line with our expectations.</p>
<p>In the NEC condition, gaze to &#x201C;Task&#x201D; is predominant during feedback for both feedback producer and receiver. In absolute numbers, the low percentages of gaze to &#x201C;Addressee&#x201D; correspond to one single instance for PR (the 0% stands for 0.267%) by the feedback producer, and six tokens for PR and three for IS by the feedback receiver. This is also in line with our expectation that the absence of visibility reduces the use of gaze as the signalling function is not available for communicative purposes.</p>
</sec>
<sec id="sec19">
<label>5.1.2</label>
<title>Gaze duration before the feedback utterance</title>
<p><xref ref-type="fig" rid="fig5">Figure 5</xref> shows the distributions of duration for each gaze direction interval, calculated from the beginning of the interval up to the moment when turn-regulating vocal feedback signals are produced. In other words, this measure indicates how long the feedback producer or receiver maintains the same gaze direction when vocal feedback signals are produced.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Duration of directed gaze to Task (yellow) and Addressee (green) by the feedback producer (left panel) and receiver (right panel) up to the moment vocal feedback signals are uttered. Density values are displayed on the <italic>y</italic>-axis, and duration of gaze direction intervals (in seconds) is shown on the <italic>x</italic>-axis. Distributions are shown for the two turn-regulating functions of vocal feedback signals, that is, Passive Recipiency (PR) and Incipient Speakership (IS). The results are shown for both visibility conditions, eye contact (EC) on top, and non-eye-contact (NEC) on the bottom; note that a limit to the <italic>x</italic>-axis has been established for improved visualisation as the values show a flat distribution up to 120&#x202F;s.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g005.tif"/>
</fig>
<p>Different trends can be observed for the participants in the EC condition (top panels). In the case of the feedback producer, both gaze to &#x201C;Task&#x201D; and &#x201C;Addressee&#x201D; behave similarly for both feedback pragmatic functions: For PR, most data are in a range of 0&#x2013;5&#x202F;s; for IS, most data are located between 0 and 2.5&#x202F;s. This suggests that when the feedback producer has the intention to take the floor with a vocal feedback signal (IS, turn-initiating feedback), they do so quickly after a change in gaze direction (either to &#x201C;Task&#x201D; or &#x201C;Addressee&#x201D;).</p>
<p>In the case of the feedback receiver, the distribution for gaze duration to &#x201C;Addressee&#x201D; is concentrated around shorter values (0&#x2013;2.5&#x202F;s) than gaze duration to &#x201C;Task,&#x201D; for which the distribution is wider and spreads across longer duration values (0&#x2013;5&#x202F;s). This is true for both PR and IS feedback signals, indicating proportionally longer gaze duration to &#x201C;Task&#x201D; than to &#x201C;Addressee&#x201D; up to the onset of the vocal feedback production.</p>
<p>In the NEC condition, the few instances of gaze to &#x201C;Addressee&#x201D; by the feedback receiver, especially for PR, show similar values to those in the EC condition, which might be explained by the evidence reported in previous studies (see Section 1.4) that speakers perceive the head orientation of their interlocutor even in the absence of visibility. This is, however, not the case for the feedback producer, where there is no gaze to &#x201C;Addressee.&#x201D;</p>
</sec>
<sec id="sec20">
<label>5.1.3</label>
<title>Gaze shift after feedback production</title>
<p><xref ref-type="fig" rid="fig6">Figure 6</xref> shows the occurrences of gaze shifts <italic>after</italic> the production of vocal feedback. A similar trend across interlocutors and visibility conditions can be noticed: The production of turn-regulating vocal feedback does not lead to a subsequent shift in gaze direction.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Shifts in gaze direction by the feedback producer and receiver after the production of a vocal feedback signal. The proportions of gaze direction shifts are displayed as percentages on the <italic>y</italic>-axis. Exact proportions and counts of occurrences are printed on the bar plots (the latter in parenthesis) for the two categories &#x201C;No Shift&#x201D; and &#x201C;Shift&#x201D; in gaze direction, displayed on the <italic>x</italic>-axis and colour-coded. Occurrences are shown for the two turn-regulating functions of vocal feedback signals, that is, Passive Recipiency (PR) and Incipient Speakership (IS), indicated on the top, and across visibility conditions, i.e., eye-contact (EC) and non-eye-contact (NEC), indicated on the right.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g006.tif"/>
</fig>
<p>A slight difference can be observed between the feedback producer and recipient in the EC condition. For the feedback producer, there is a shift in gaze direction in only 9 and 8% of the cases for PR and IS, respectively. For PR, these few cases mostly imply shifting the gaze from the task towards the feedback receiver (12 cases, possibly confirming the availability as a listener). For IS, there is an equal number of shifts towards the task and the feedback receiver (four cases each). A shift in gaze direction after the production of vocal feedback by the receiver occurs in 20% of PR and 15% of IS vocal feedback signals, which is more than what we observe for the feedback producer. For PR, the gaze shift is almost exclusively from addressee to the task (32 cases). For IS, there is the same prevalence of shifts from &#x201C;Addressee&#x201D; towards &#x201C;Task&#x201D; (10 cases and 5 from task to addressee). In other words, the few cases in which vocal feedback is associated with a shift in gaze direction by the primary speaker occur when the primary speaker looks at the addressee before receiving feedback, and back at the task after ensuring that the listener is following the conversation.</p>
<p>However, overall, there are very few data points of gaze shift after a vocal feedback signal across the two visibility conditions. Therefore, it appears that vocal feedback signals rarely occur with changes in the gaze direction of the interlocutors.</p>
</sec>
</sec>
<sec id="sec21">
<label>5.2</label>
<title>Overall gaze direction intervals and their duration</title>
<p>We found 1,649 gaze direction intervals in the eye-contact condition (EC; 80% of the total) and 354 gaze direction intervals in the non-eye-contact condition (NEC; 20% of the total), showing that, in line with our expectations, participants change the direction of their gaze far less often when eye contact is inhibited.</p>
<p>However, the relative proportions of gaze direction intervals directed to &#x201C;Task&#x201D; or &#x201C;Addressee&#x201D; displayed in <xref ref-type="fig" rid="fig7">Figure 7</xref> (<italic>y</italic>-axis) do not differ greatly across conditions. In the EC condition, there is an almost equal proportion of gaze occurrences to &#x201C;Task&#x201D; and &#x201C;Addressee&#x201D; (50.5 and 49.5% respectively). In the NEC condition, where the interlocutor is behind an opaque panel, the proportion of gaze to &#x201C;Task&#x201D; increases only slightly (60.7%) as compared to the EC condition. On the other hand, the proportion of gaze to &#x201C;Addressee&#x201D; (39.3%), although lower than in the EC condition, indicates that participants gaze away from the task quite often, looking at the panel, that is, in the direction of the interlocutor&#x2019;s voice, namely, where the addressee is known to be sitting.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Occurrences of gaze direction intervals. Proportions of gaze direction intervals are displayed in percentages on the <italic>y</italic>-axis. The two gaze directions are colour-coded as in the legend: gaze to task (Task) in yellow and gaze towards addressee (Addressee) in green. The exact proportions and counts of occurrences are printed on the bar plots (the latter in parentheses) for the two gaze directions displayed on the <italic>x</italic>-axis. The two conditions, eye contact (EC) and non-eye contact (NEC), are shown in the boxes above the bars.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g007.tif"/>
</fig>
<p>Mean gaze duration (<xref ref-type="fig" rid="fig8">Figure 8</xref>) shows that the time spent looking at the &#x201C;Task&#x201D; is longer than that spent looking towards the &#x201C;Addressee&#x201D; in both conditions (4.75&#x202F;s in EC and 18.2&#x202F;s in NEC). However, the time spent looking at the &#x201C;Task&#x201D; shows high variability and can potentially be longer, especially when eye contact is inhibited (see large error bars especially for NEC). There is less variability in the time spent looking towards the &#x201C;Addressee,&#x201D; with a mean duration of 1.55&#x202F;s in the EC condition and 0.74&#x202F;s in the NEC condition, especially in the NEC condition (small error bars).</p>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>Mean duration of gaze direction intervals. The duration in seconds is shown on the <italic>y</italic>-axis, while the exact mean value is printed above the bar plots. The grey lines represent standard errors. The two categories, gaze to task (Task) and gaze towards addressee (Addressee), are displayed on the <italic>x</italic>-axis. The two conditions, eye-contact (EC) and non-eye-contact (NEC), are shown in the boxes above the bars.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g008.tif"/>
</fig>
<p>In line with our expectations, these two metrics taken together show that the lower number of gaze direction intervals in the NEC condition could be explained by the fact that participants gaze for longer periods at the task than in the EC condition, leaving little room for switching gaze direction.</p>
</sec>
<sec id="sec22">
<label>5.3</label>
<title>Overall vocal feedback rate</title>
<p>Finally, <xref ref-type="fig" rid="fig9">Figure 9</xref> shows the rate of vocal feedback production across pragmatic functions and visibility conditions. These values indicate that the vocal feedback rate is higher in the NEC than in the EC condition, as expected. Across feedback functions, our analysis reveals a proportionally higher rate (almost double) of PR than IS vocal feedback expressions.</p>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>Turn-regulating vocal feedback rate across eye contact (EC) and non-eye-contact (NEC) conditions. The rate, operationalised as feedback signals per minute, is displayed on the y-axis. The mean rate is printed on the bar plots. The two turn-regulating functions of the feedback signals, Passive Recipiency and Incipient Speakership, are displayed on the <italic>x</italic>-axis.</p>
</caption>
<graphic xlink:href="fcomm-10-1524453-g009.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="sec23">
<label>6</label>
<title>Discussion</title>
<p>Our first research question (RQ 1 in Section 2) concerned the interplay between gaze and vocal turn-regulating feedback by the feedback producer and receiver in terms of the number of gaze direction intervals, their duration, and the shift of gaze direction after the production of vocal feedback.</p>
<p>Our results show that, in a face-to-face task-based interaction, the feedback producer (the secondary speaker, RQ 1a) almost exclusively looks at the task while producing vocal feedback, independent of the intention to take the floor. This result suggests that feedback producers might rely on cues other than eye gaze to signal the floor management function of vocal feedback. For example, they may use prosody (in particular, intonational cues), which has been found to vary according to the turn-regulating function and type of feedback (<xref ref-type="bibr" rid="ref82">Savino, 2010</xref>, <xref ref-type="bibr" rid="ref83">2011</xref>, <xref ref-type="bibr" rid="ref84">2012</xref>; <xref ref-type="bibr" rid="ref86">Savino and Refice, 2013</xref>; <xref ref-type="bibr" rid="ref87">Sbranna et al., 2022</xref>; <xref ref-type="bibr" rid="ref88">Sbranna et al., 2024</xref>).</p>
<p>The results for gaze behaviour of the feedback receiver (the primary speaker, RQ 1b) are different, with a higher percentage of gaze towards the Addressee for both feedback functions, but especially for IS (when the secondary speaker intends to take the floor). The latter case is in line with the literature, in which the primary speaker is said to gaze at the interlocutor when signalling a speaker change. Since interactions included a visual task, it is understandable that we find a higher level of gaze to task than predicted by previous studies on gaze and turn alternation, which were mostly based on free conversations and reported mutual gaze in correspondence with feedback. Nevertheless, our data show a trend in the expected direction for the feedback receiver: gaze to task, showing unavailability for a turn transition, mostly prompts continuers (PR), whereas gaze to addressee prompts a high percentage of turn-initial feedback (IS). Comparing these results with those obtained by <xref ref-type="bibr" rid="ref92">Spaniol et al. (2023)</xref>, we can confirm the reported prevalence of averted gaze together with vocal feedback by the feedback producer. However, in our dataset, we also find that the feedback receiver uses a greater amount of gaze directed towards the interlocutor when the interlocutor is producing vocal feedback. In relative terms across functions, gaze is produced less with PR and more with IS, showing that the signalling function of gaze is only marginally used to elicit feedback and is more often used to regulate turns. This more fine-grained description of gaze, going beyond the binary distinction between &#x2018;mutual&#x2019; and &#x2018;averted&#x2019; gaze is a possible explanation for the difference between our results and previous ones.</p>
<p>We also investigated the duration of the gaze direction intervals up to the moment when vocal feedback signals are uttered (RQ 1c) to explore the temporal details (as in <xref ref-type="bibr" rid="ref59">Kendrick et al., 2023</xref> and <xref ref-type="bibr" rid="ref71">Ond&#x00E1;&#x0161; et al., 2023</xref>) of the time window of gaze under which the vocal feedback is uttered. We observe a similar range of duration values as <xref ref-type="bibr" rid="ref71">Ond&#x00E1;&#x0161; et al. (2023)</xref> for the feedback receiver, who obtains a quick vocal response after initiating gaze, independent of the feedback turn-regulating function. This suggests that looking at the secondary speaker may be primarily interpreted by the latter as a feedback request. The findings on these temporal details can also be related to the concept of an advantageous mutual gaze window for feedback insertion proposed in previous studies (<xref ref-type="bibr" rid="ref11">Bavelas et al., 2012</xref>; <xref ref-type="bibr" rid="ref10">Bavelas et al., 2002</xref>). From our dataset, it emerges that an advantageous gaze window is not necessarily mutual&#x2014;since the occurrences of gaze to &#x201C;Addressee&#x201D; while uttering feedback, i.e., by the secondary speaker, are much lower than those of the primary speaker&#x2014;but can be <italic>unilateral</italic> as even the sole activation of gaze by the primary speakers elicits a quick vocal response. Instead, for the feedback producer, the data suggest a tendency for shorter gaze duration (either to &#x201C;Task&#x201D; or &#x201C;Addressee&#x201D;) before turn-initiating feedback signals, i.e., the feedback producer quickly takes the floor with a vocal feedback signal after a change in their gaze direction. This result contrasts with the findings of <xref ref-type="bibr" rid="ref59">Kendrick et al. (2023)</xref>, who claimed that the direction of speaker gaze does not affect the speed of general turn transitions. The secondary speaker might consider moments of perturbations of the previous state of balance in the conversation, that is, in proximity to a change in activation/deactivation of visual cues from their side, as better suited for a change in speakership (see Complex Dynamic System Theory, <xref ref-type="bibr" rid="ref21">Cameron and Larsen-Freeman, 2007</xref>).</p>
<p>Finally, we find no evidence that the production of turn-regulating vocal feedback causes a shift in gaze direction by either the feedback producer or receiver in most cases (RQ 1d), as found by <xref ref-type="bibr" rid="ref10">Bavelas et al. (2002)</xref>. This result might also be interpreted in the light of our task-based setting, in which visual attention to the task materials was necessary to complete the task: The overall longer gaze time to task during the dialogues increases the probability that vocal feedback (at both their onset and offset) corresponds to these gaze intervals. However, in the very few cases in which a gaze shift occurs, it is prevalent in the feedback receiver after PR signals, which is in line with the findings of <xref ref-type="bibr" rid="ref10">Bavelas et al. (2002)</xref>. Similar to their findings, these shifts are from gaze to addressee towards the task, showing that the visual contact established by the primary speaker might indeed have the function of eliciting vocal feedback as it is concluded right after its utterance.</p>
<p>Our second research question (RQ 2 as in Section 2) concerned the effect of non-visibility between participants on their use of gaze and vocal feedback.</p>
<p>One striking remark on overall gaze behaviour is that, despite the presence of a visual competitor in this experimental setting, participants still show high relative proportions of occurrences of gaze directed towards the interlocutor, independent of whether eye contact was inhibited or not (in line with <xref ref-type="bibr" rid="ref32">Edlund et al., 2012</xref>; <xref ref-type="bibr" rid="ref55">Kato et al., 2010</xref>; <xref ref-type="bibr" rid="ref67">Nakano et al., 2008</xref> about the ability of speakers to identify interlocutors&#x2019; head direction in darkness, based on acoustic cues). The greatest difference in overall gaze behaviour across visibility conditions was related to the length of these gaze direction intervals, with participants spending more time looking at the task when visual contact was impeded (RQ 2a). This result is in line with previous studies reporting that non-visibility between interlocutors in a conversation causes a drastic drop in the search for visual contact (<xref ref-type="bibr" rid="ref6">Argyle et al., 1973</xref>; <xref ref-type="bibr" rid="ref75">Renklint et al., 2012</xref>). This might be explained by the fact that in the EC condition, gaze fulfils a communicative signalling function in addition to its primary sensing function, whereas in the NEC condition only the sensing function is possible&#x2014;we impeded eye contact, but participants could still sense their environment&#x2014;reducing the occasions in which switching gaze direction is necessary. This result, while seemingly trivial at first, opens up key future questions as to which functions eye gaze fulfil and how these functions are distributed when eye contact is impeded. Since we found that the difference in the proportion of occurrences of gaze to addressee is only 10% across the two conditions, the question of which speech events relate to these short but still represented occurrences of gazes to addressee in the absence of visual contact remains open as no correspondence with turn-regulating feedback (2b) was found. One speculation we propose is that eye gaze directed towards the sound source in non-visibility conditions may be activated at moments of heightened attention, serving a focusing function on the source of information in response to increased cognitive demand.</p>
<p>Vocal feedback rate shows that the unavailability of the visual channel boosts the production of vocal feedback signals (RQ 2c), which is in line with previous findings of increased use of backchannels and vocal resources in general in the absence of eye contact (e.g., <xref ref-type="bibr" rid="ref25">Cook and Lalljee, 1972</xref>; <xref ref-type="bibr" rid="ref19">Boyle et al., 1994</xref>; <xref ref-type="bibr" rid="ref68">Neiberg and Gustafson, 2011</xref>). We also found an overall prevalence of PR over IS vocal feedback, which has also been reported in previous task-based studies on vocal feedback in Tangram game-based dyadic conversations in German (<xref ref-type="bibr" rid="ref92">Spaniol et al., 2023</xref>) and Map Tasks (<xref ref-type="bibr" rid="ref3">Anderson et al., 1991</xref>) in Italian and German (<xref ref-type="bibr" rid="ref82">Savino, 2010</xref>, <xref ref-type="bibr" rid="ref83">2011</xref>, <xref ref-type="bibr" rid="ref84">2012</xref>; <xref ref-type="bibr" rid="ref86">Savino and Refice, 2013</xref>; <xref ref-type="bibr" rid="ref88">Sbranna et al., 2024</xref>; <xref ref-type="bibr" rid="ref97">Wehrle, 2023</xref>; <xref ref-type="bibr" rid="ref87">Sbranna et al., 2022</xref>). Thus, this phenomenon might be related to a collaborative-task-based setting, in which speakers tend to alternately lead the conversation, leaving the interlocutor the role of acknowledging the other&#x2019;s speech to ensure that the information necessary to complete the task has been successfully received, that is, to update the common ground/knowledge (<xref ref-type="bibr" rid="ref24">Clark and Schaefer, 1989</xref>).</p>
</sec>
<sec sec-type="conclusions" id="sec24">
<label>7</label>
<title>Conclusion</title>
<p>We conducted a small-sample study on the interplay between gaze and turn-regulating vocal feedback, with a 2-fold goal. Using video-recorded Tangram game conversations by six dyads of Italian speakers in eye-contact and non-eye-contact conditions, we analysed (1) the relationship between turn-regulating vocal feedback&#x2014;acknowledgements with the functions of &#x201C;Incipient Speakership&#x201D; (turn-initiating) and &#x201C;Passive Recipiency&#x201D; (continuers)&#x2014;and eye gaze directions (to the Task or the Addressee) and (2) how far this relation, as well as overall gaze and feedback behaviour, change when eye contact is impeded. In the following paragraphs, we address the limitations and innovations of the study.</p>
<p>First, this analysis was based on a relatively limited number of participants. Increasing the sample size would allow for robust statistical testing of these preliminary findings, and therefore, for a certain degree of generalisability of the phenomena reported here. Most previous studies on gaze and turn alternation were based on free conversation, whereas our task-based design was meant to match a comparable study design on gaze and turn-regulating feedback. A task-based setting should not be regarded as less generalisable to real-world situations than free conversations, since many real-world dialogues involve a visual competitor, that is, an object in the immediate environment that is also the topic of conversation. These conversational contexts are even more challenging for interlocutors as they have to compromise between looking at the object and using gaze and speech to effectively manage the conversation. However, to guarantee full comparability with previous studies and partial out any variation from previous studies due to different designs, a replication of this study with free conversations, including an analysis of eye gaze at turn transition for comparison, would be beneficial. This would shed light on the differences and similarities between eye gaze during turn alternation and turn-regulating vocal feedback. Moreover, such an analysis should account for the content of the sentences involved in the turn alternations (e.g., adjacency pairs) as gaze behaviour may vary depending on pragmatic aspects.</p>
<p>Second, we did not use eye tracking technology, meaning that our annotations were made from the observer&#x2019;s perspective, which may differ from the participant&#x2019;s perspective elicited in an eye tracking study. For our research goal, this did not represent a problem as participants were video recorded frontally and eye movements were fully detectable. However, the complementary use of eye-tracking techniques in the future, apart from allowing automated analysis of larger datasets, could provide a higher level of detail, especially useful for further investigating the more variable nature of the addressee-oriented gaze we found in the absence of eye contact.</p>
<p>Finally, our investigation is limited to the relationship between eye gaze and turn-regulating vocal feedback, but expanding the current study to visual types of feedback such as head nods and including prosodic aspects of feedback would provide a more complete picture.</p>
<p>Despite these shortcomings, we provided insightful preliminary findings regarding the under-researched relationship between eye gaze and turn-regulating feedback. As compared to previous designs based on a binary distinction of &#x2018;mutual&#x2019; vs. &#x2018;averted&#x2019; gaze, we proposed a tripartite distinction including unilateral directed gaze, and analysing gaze by the feedback producer and receiver, which provided novel insights. In particular, the finding that the feedback receiver (the primary speaker) uses gaze more often than the feedback producer (the secondary speaker) suggests that <italic>unilateral</italic> gaze (i.e., when gaze by one participant is not reciprocated by the interlocutor) is also used as a successful and efficient feedback request. Furthermore, we provided new findings about the interplay between vocal feedback and gaze behaviour in a task-based context in the absence of eye contact, which, to our knowledge, had not been investigated before. We found that gaze is directed towards the interlocutor even in the absence of eye contact, but not for turn-regulating vocal feedback, opening new research perspectives. Future research examining unilateral and mutual gaze will help confirm and expand these findings to provide a more complete picture of these interactional phenomena.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec25">
<title>Data availability statement</title>
<p>The data table and analysis code presented in this study can be found at <ext-link xlink:href="https://osf.io/3dqzk/" ext-link-type="uri">https://osf.io/3dqzk/</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="sec26">
<title>Ethics statement</title>
<p>Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec sec-type="author-contributions" id="sec27">
<title>Author contributions</title>
<p>SS: Conceptualization, Data curation, Formal analysis, Methodology, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. MS: Methodology, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing, Data curation, Investigation, Resources. FB: Methodology, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. MG: Methodology, Supervision, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec28">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by the Cluster Development Program Language Challenges, a funding line within the Excellent Research Support Program of the University of Cologne, and the German Research Foundation (DFG), grant number 281511265, SFB 1252 Prominence in Language.</p>
</sec>
<sec sec-type="COI-statement" id="sec29">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec30">
<title>Generative AI statement</title>
<p>The authors declare that no Gen AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="sec31">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup>Please note that by using the opaque panel in the non-visibility condition, we restrict the amount of information that can be sampled from the environment, but not the sensing function itself, as was done in previous studies comparing light vs. darkness conditions (<xref ref-type="bibr" rid="ref75">Renklint et al., 2012</xref> as mentioned in Section 2.4).</p></fn>
<fn id="fn0002"><p><sup>2</sup>Participants alternated their role as Director or Matcher in each round, so that the distribution of role type was balanced between partners across the whole recording session, i.e., each speaker of a dyad played the role of Director 11 times and that of Matcher 11 times.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alibali</surname> <given-names>M. W.</given-names></name> <name><surname>Heath</surname> <given-names>D. C.</given-names></name> <name><surname>Myers</surname> <given-names>H. J.</given-names></name></person-group> (<year>2001</year>). <article-title>Effects of visibility between speaker and listener on gesture production: some gestures are meant to be seen</article-title>. <source>J. Mem. Lang.</source> <volume>44</volume>, <fpage>169</fpage>&#x2013;<lpage>188</lpage>. doi: <pub-id pub-id-type="doi">10.1006/jmla.2000.2752</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Amador-Moreno</surname> <given-names>C. P.</given-names></name> <name><surname>McCarthy</surname> <given-names>M.</given-names></name> <name><surname>O&#x2019;Keeffe</surname> <given-names>A.</given-names></name></person-group> (<year>2013</year>). &#x201C;<article-title>Can English provide a framework for Spanish response tokens?</article-title>&#x201D; in <source>Yearbook of Corpus Linguistics and Pragmatics 2013: New Domains and Methodologies</source> (<publisher-loc>Dordrecht</publisher-loc>: <publisher-name>Springer Netherlands</publisher-name>), <fpage>175</fpage>&#x2013;<lpage>201</lpage>.</citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Anderson</surname> <given-names>A. H.</given-names></name> <name><surname>Bader</surname> <given-names>M.</given-names></name> <name><surname>Bard</surname> <given-names>E. G.</given-names></name> <name><surname>Boyle</surname> <given-names>E.</given-names></name> <name><surname>Doherty</surname> <given-names>G.</given-names></name> <name><surname>Garrod</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>1991</year>). <article-title>The HCRC map task corpus</article-title>. <source>Lang. Speech</source> <volume>34</volume>, <fpage>351</fpage>&#x2013;<lpage>366</lpage>. doi: <pub-id pub-id-type="doi">10.1177/002383099103400404</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Argyle</surname> <given-names>M.</given-names></name> <name><surname>Cook</surname> <given-names>M.</given-names></name></person-group> (<year>1976</year>). <source>Gaze and Mutual Gaze</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Argyle</surname> <given-names>M.</given-names></name> <name><surname>Dean</surname> <given-names>J.</given-names></name></person-group> (<year>1965</year>). <article-title>Eye-contact, distance and affiliation</article-title>. <source>Sociometry</source> <volume>28</volume>:<fpage>289</fpage>. doi: <pub-id pub-id-type="doi">10.2307/2786027</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Argyle</surname> <given-names>M.</given-names></name> <name><surname>Ingham</surname> <given-names>R.</given-names></name> <name><surname>Alkema</surname> <given-names>F.</given-names></name> <name><surname>McCallin</surname> <given-names>M.</given-names></name></person-group> (<year>1973</year>). <article-title>The different functions of gaze</article-title>. <source>Semiotica</source> <volume>7</volume>, <fpage>19</fpage>&#x2013;<lpage>32</lpage>. doi: <pub-id pub-id-type="doi">10.1515/semi.1973.7.1.19</pub-id>, PMID: <pub-id pub-id-type="pmid">40181836</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Argyle</surname> <given-names>M.</given-names></name> <name><surname>Lalljee</surname> <given-names>M.</given-names></name> <name><surname>Cook</surname> <given-names>M.</given-names></name></person-group> (<year>1968</year>). <article-title>The effects of visibility on interaction in a dyad</article-title>. <source>Human relations</source> <volume>21</volume>, <fpage>3</fpage>&#x2013;<lpage>17</lpage>.</citation></ref>
<ref id="ref8"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Auer</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Gaze, addressee selection and turn-taking in three-party interaction</article-title>,&#x201D; in <source>Eye-tracking in interaction: Studies on the role of eye gaze in dialogue</source>. eds. <person-group person-group-type="editor"><name><surname>Br&#x00F4;ne</surname> <given-names>G.</given-names></name> <name><surname>Oben</surname> <given-names>B.</given-names></name></person-group> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>John Benjamins</publisher-name>), <fpage>197</fpage>&#x2013;<lpage>232</lpage>.</citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bavelas</surname> <given-names>J. B.</given-names></name> <name><surname>Chovil</surname> <given-names>N.</given-names></name> <name><surname>Lawrie</surname> <given-names>D. A.</given-names></name> <name><surname>Wade</surname> <given-names>A.</given-names></name></person-group> (<year>1992</year>). <article-title>Interactive gestures</article-title>. <source>Discourse Process.</source> <volume>15</volume>, <fpage>469</fpage>&#x2013;<lpage>489</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01638539209544823</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bavelas</surname> <given-names>J. B.</given-names></name> <name><surname>Coates</surname> <given-names>L.</given-names></name> <name><surname>Johnson</surname> <given-names>T.</given-names></name></person-group> (<year>2002</year>). <article-title>Listener responses as a collaborative process: the role of gaze</article-title>. <source>J. Commun.</source> <volume>52</volume>, <fpage>566</fpage>&#x2013;<lpage>580</lpage>. doi: <pub-id pub-id-type="doi">10.1111/jcom.2002.52.issue-3</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Bavelas</surname> <given-names>J. B.</given-names></name> <name><surname>De Jong</surname> <given-names>P.</given-names></name> <name><surname>Korman</surname> <given-names>H.</given-names></name> <name><surname>Jordan</surname> <given-names>S. S.</given-names></name></person-group> (<year>2012</year>). <conf-name>Beyond back-channels: A three-step model of grounding in face-to-face dialogue. In Proceedings of Interdisciplinary Workshop on Feedback Behaviors in Dialog</conf-name>, <fpage>5</fpage>&#x2013;<lpage>6</lpage>.</citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bavelas</surname> <given-names>J. B.</given-names></name> <name><surname>Gerwing</surname> <given-names>J.</given-names></name> <name><surname>Sutton</surname> <given-names>C.</given-names></name> <name><surname>Prevost</surname> <given-names>D.</given-names></name></person-group> (<year>2008</year>). <article-title>Gesturing on the telephone: independent effects of dialogue and visibility</article-title>. <source>J. Mem. Lang.</source> <volume>58</volume>, <fpage>495</fpage>&#x2013;<lpage>520</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jml.2007.02.004</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Beattie</surname> <given-names>G. W.</given-names></name></person-group> (<year>1978</year>). <article-title>Floor apportionment and gaze in conversational dyads</article-title>. <source>British J. Soc. Clinic. Psychol.</source> <volume>17</volume>, <fpage>7</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2044-8260.1978.tb00889.x</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Beattie</surname> <given-names>G. W.</given-names></name></person-group> (<year>1979</year>). <article-title>Contextual constraints on the floor-apportionment function of speaker-gaze in dyadic conversations</article-title>. <source>Br. J. Soc. Clin. Psychol.</source> <volume>18</volume>, <fpage>391</fpage>&#x2013;<lpage>392</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2044-8260.1979.tb00909.x</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Bertrand</surname> <given-names>R.</given-names></name> <name><surname>Ferr&#x00E9;</surname> <given-names>G.</given-names></name> <name><surname>Blache</surname> <given-names>P.</given-names></name> <name><surname>Espesser</surname> <given-names>R.</given-names></name> <name><surname>Rauzy</surname> <given-names>S.</given-names></name></person-group> (<year>2007</year>). &#x201C;<article-title>Backchannels revisited from a multimodal perspective</article-title>,&#x201D; in <source>Proceedings of the ISCA Workshop &#x201C;Auditory Audio-Visual Processing&#x201D;</source>. Available at: <ext-link xlink:href="https://www.isca-archive.org/avsp_2007/bertrand07_avsp.pdf" ext-link-type="uri">https://www.isca-archive.org/avsp_2007/bertrand07_avsp.pdf</ext-link>.</citation></ref>
<ref id="ref16"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Bissonnette</surname> <given-names>V. L.</given-names></name></person-group> (<year>1993</year>). <source>Interdependence in dyadic gazing [doctoral dissertation, the University of Texas at Arlington]</source>. <publisher-loc>Ann Arbour</publisher-loc>: <publisher-name>ProQuest Dissertations Publishing</publisher-name>.</citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blythe</surname> <given-names>J.</given-names></name> <name><surname>Gardner</surname> <given-names>R.</given-names></name> <name><surname>Mushin</surname> <given-names>I.</given-names></name> <name><surname>Stirling</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Tools of engagement: selecting a next speaker in australian aboriginal multiparty conversations</article-title>. <source>Res. Lang. Soc. Interact.</source> <volume>51</volume>, <fpage>145</fpage>&#x2013;<lpage>170</lpage>. doi: <pub-id pub-id-type="doi">10.1080/08351813.2018.1449441</pub-id></citation></ref>
<ref id="ref0020"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boersma</surname> <given-names>P.</given-names></name> <name><surname>Weenink</surname> <given-names>D.</given-names></name></person-group> (<year>2001</year>). <article-title>Praat, a system for doing phonetics by computer</article-title>. <source>Glot International</source> <volume>5</volume>, <fpage>341</fpage>&#x2013;<lpage>345</lpage>.</citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Borr&#x00E0;s-Comes</surname> <given-names>J.</given-names></name> <name><surname>Kaland</surname> <given-names>C.</given-names></name> <name><surname>Prieto</surname> <given-names>P.</given-names></name> <name><surname>Swerts</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Audiovisual correlates of interrogativity: a comparative analysis of Catalan and Dutch</article-title>. <source>J. Nonverbal Behav.</source> <volume>38</volume>, <fpage>53</fpage>&#x2013;<lpage>66</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10919-013-0162-0</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boyle</surname> <given-names>E.</given-names></name> <name><surname>Anderson</surname> <given-names>A.</given-names></name> <name><surname>Newlands</surname> <given-names>A.</given-names></name></person-group> (<year>1994</year>). <article-title>The effects of visibility on dialogue and performance in a cooperative problem-solving task</article-title>. <source>Lang. Speech</source> <volume>37</volume>:<fpage>l</fpage>.</citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Br&#x00F4;ne</surname> <given-names>G.</given-names></name> <name><surname>Oben</surname> <given-names>B.</given-names></name> <name><surname>Jehoul</surname> <given-names>A.</given-names></name> <name><surname>Vranjes</surname> <given-names>J.</given-names></name> <name><surname>Feyaerts</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>Eye gaze and viewpoint in multimodal interaction management</article-title>. <source>Cog. Linguis.</source> <volume>28</volume>, <fpage>449</fpage>&#x2013;<lpage>483</lpage>. doi: <pub-id pub-id-type="doi">10.1515/cog-2016-0119</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cameron</surname> <given-names>L.</given-names></name> <name><surname>Larsen-Freeman</surname> <given-names>D.</given-names></name></person-group> (<year>2007</year>). <article-title>Complex systems and applied linguistics</article-title>. <source>Int. J. Appl. Linguist.</source> <volume>17</volume>, <fpage>226</fpage>&#x2013;<lpage>240</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1473-4192.2007.00148.x</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ca&#x00F1;igueral</surname> <given-names>R.</given-names></name> <name><surname>Hamilton</surname> <given-names>A. F. C.</given-names></name></person-group> (<year>2019</year>). <article-title>The role of eye gaze during natural social interactions in typical and autistic people</article-title>. <source>Front. Psychol.</source> <volume>10</volume>:<fpage>560</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2019.00560</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clancy</surname> <given-names>P. M.</given-names></name> <name><surname>Thompson</surname> <given-names>S. A.</given-names></name> <name><surname>Suzuki</surname> <given-names>R.</given-names></name> <name><surname>Tao</surname> <given-names>H.</given-names></name></person-group> (<year>1996</year>). <article-title>The conversational use of reactive tokens in English, Japanese, and Mandarin</article-title>. <source>J. Pragmat.</source> <volume>26</volume>, <fpage>355</fpage>&#x2013;<lpage>387</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0378-2166(95)00036-4</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>H. H.</given-names></name> <name><surname>Schaefer</surname> <given-names>E. F.</given-names></name></person-group> (<year>1989</year>). <article-title>Contributing to discourse</article-title>. <source>Cogn. Sci.</source> <volume>13</volume>, <fpage>259</fpage>&#x2013;<lpage>294</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15516709cog1302_7</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cook</surname> <given-names>M.</given-names></name> <name><surname>Lalljee</surname> <given-names>M. G.</given-names></name></person-group> (<year>1972</year>). <article-title>Verbal substitutes for visual signals in interaction</article-title>. <source>Semiotica</source> <volume>6</volume>, <fpage>212</fpage>&#x2013;<lpage>221</lpage>. doi: <pub-id pub-id-type="doi">10.1515/semi.1972.6.3.212</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cummins</surname> <given-names>F.</given-names></name></person-group> (<year>2012</year>). <article-title>Gaze and blinking in dyadic conversation: a study in coordinated behaviour among individuals</article-title>. <source>Lang. Cog. Proc.</source> <volume>27</volume>, <fpage>1525</fpage>&#x2013;<lpage>1549</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01690965.2011.615220</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Degutyte</surname> <given-names>Z.</given-names></name> <name><surname>Astell</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>The role of eye gaze in regulating turn taking in conversations: a systematized review of methods and findings</article-title>. <source>Front. Psychol.</source> <volume>12</volume>:<fpage>616471</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2021.616471</pub-id>, PMID: <pub-id pub-id-type="pmid">33897526</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Drummond</surname> <given-names>K.</given-names></name> <name><surname>Hopper</surname> <given-names>R.</given-names></name></person-group> (<year>1993</year>). <article-title>Back channels revisited: acknowledgment tokens and speakership incipiency</article-title>. <source>Res. Lang. Soc. Interact.</source> <volume>26</volume>, <fpage>157</fpage>&#x2013;<lpage>177</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15327973rlsi2602_3</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duncan</surname> <given-names>S.</given-names></name></person-group> (<year>1972</year>). <article-title>Some signals and rules for taking speaking turns in conversations</article-title>. <source>J. Personal. Soc. Psychol.</source> <volume>23</volume>, <fpage>283</fpage>&#x2013;<lpage>292</lpage>. doi: <pub-id pub-id-type="doi">10.1037/h0033031</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Duncan</surname> <given-names>S.</given-names></name> <name><surname>Fiske</surname> <given-names>D. W.</given-names></name></person-group> (<year>1977</year>). <source>Face-to-face interaction: Research, methods, and theory</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Routledge</publisher-name>.</citation></ref>
<ref id="ref31"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Eberhard</surname> <given-names>K.M.</given-names></name> <name><surname>Nicholson</surname> <given-names>H</given-names></name></person-group>. (<year>2010</year>). <source>Coordination of understanding in face-to-face narrative dialogue</source>. In <conf-name>Proceedings of the Annual Meeting of the Cognitive Science Society</conf-name></citation></ref>
<ref id="ref32"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Edlund</surname> <given-names>J.</given-names></name> <name><surname>Heldner</surname> <given-names>M.</given-names></name> <name><surname>Gustafson</surname> <given-names>J</given-names></name></person-group>. (<year>2012</year>). &#x201C;<article-title>On the effect of the acoustic environment on the accuracy of perception of speaker orientation from auditory cues alone</article-title>,&#x201D; in <source>The Proceedings of 13th Annual Conference of the International Speech Communication Association</source>. pp. <fpage>1482</fpage>&#x2013;<lpage>1485</lpage>.</citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Egbert</surname> <given-names>M. M.</given-names></name></person-group> (<year>1996</year>). <article-title>Context-sensitivity in conversation: Eye gaze and the German repair initiator bitte?</article-title> <source>Language in Society</source> <volume>25</volume>, <fpage>587</fpage>&#x2013;<lpage>612</lpage>.</citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fellegy</surname> <given-names>A. M.</given-names></name></person-group> (<year>1995</year>). <article-title>Patterns and functions of minimal response</article-title>. <source>Amer. Speech Int. J. Educ. Best Prac.</source> <volume>70</volume>, <fpage>186</fpage>&#x2013;<lpage>199</lpage>. doi: <pub-id pub-id-type="doi">10.2307/455815</pub-id>, PMID: <pub-id pub-id-type="pmid">39964225</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Ferr&#x00E9;</surname> <given-names>G.</given-names></name> <name><surname>Renaudier</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>Unimodal and bimodal backchannels in conversational English</article-title>. <conf-name>In Proceedings of SemDial, Saarbr&#x00FC;cken, Germany</conf-name>, <volume>27&#x2013;37</volume>.</citation></ref>
<ref id="ref36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Foulsham</surname> <given-names>T.</given-names></name> <name><surname>Cheng</surname> <given-names>J. T.</given-names></name> <name><surname>Tracy</surname> <given-names>J. L.</given-names></name> <name><surname>Henrich</surname> <given-names>J.</given-names></name> <name><surname>Kingstone</surname> <given-names>A.</given-names></name></person-group> (<year>2010</year>). <article-title>Gaze allocation in a dynamic situation: effects of social status and speaking</article-title>. <source>Cognition</source> <volume>117</volume>, <fpage>319</fpage>&#x2013;<lpage>331</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cognition.2010.09.003</pub-id></citation></ref>
<ref id="ref37"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Fries</surname> <given-names>C. C.</given-names></name></person-group> (<year>1952</year>). <source>The structure of English</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Longmans, Green; Company</publisher-name>.</citation></ref>
<ref id="ref38"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Gardner</surname> <given-names>R.</given-names></name></person-group> (<year>2001</year>). <source>When listeners talk: Response tokens and listener stance</source>. <publisher-loc>Amsterdam</publisher-loc>: <publisher-name>John Benjamins Publishing Company</publisher-name>.</citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gobel</surname> <given-names>M. S.</given-names></name> <name><surname>Kim</surname> <given-names>H. S.</given-names></name> <name><surname>Richardson</surname> <given-names>D. C.</given-names></name></person-group> (<year>2015</year>). <article-title>The dual function of social gaze</article-title>. <source>Cognition</source> <volume>136</volume>, <fpage>359</fpage>&#x2013;<lpage>364</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cognition.2014.11.040</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goodwin</surname> <given-names>C.</given-names></name></person-group> (<year>1980</year>). <article-title>Restarts, pauses, and the achievement of a state of mutual gaze at turn-beginning</article-title>. <source>Sociol. Inquiry</source> <volume>5</volume>, <fpage>272</fpage>&#x2013;<lpage>302</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1475-682X.1980.tb00023.x</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Goodwin</surname> <given-names>C.</given-names></name></person-group> (<year>1981</year>). <source>Conversational organization: Interaction between speakers and hearers</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Academic Press</publisher-name>.</citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harrigan</surname> <given-names>J. A.</given-names></name> <name><surname>Steffen</surname> <given-names>J. J.</given-names></name></person-group> (<year>1983</year>). <article-title>Gaze as a turn-exchange signal in group conversations</article-title>. <source>Br. J. Soc. Psychol.</source> <volume>22</volume>:<fpage>167&#x2013;168</fpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2044-8309.1983.tb00578.x</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Hasegawa</surname> <given-names>Y.</given-names></name></person-group> (<year>2014</year>). <source>Japanese: A linguistic introduction</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation></ref>
<ref id="ref44"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Heldner</surname> <given-names>M.</given-names></name> <name><surname>Hjalmarsson</surname> <given-names>A.</given-names></name> <name><surname>Edlund</surname> <given-names>J.</given-names></name></person-group> (<year>2013</year>). <article-title>Backchannel relevance spaces</article-title>. In <person-group person-group-type="editor"><name><surname>Asu</surname> <given-names>E. L.</given-names></name> <name><surname>Lippus</surname> <given-names>P.</given-names></name></person-group> (Eds.), <conf-name>Nordic Prosody: Proceedings of the XIth Conference, Tartu 2012</conf-name> (pp. <fpage>137</fpage>&#x2013;<lpage>146</lpage>).</citation></ref>
<ref id="ref45"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Heritage</surname> <given-names>J.</given-names></name></person-group> (<year>1984</year>). &#x201C;<article-title>A change-of-state token and aspects of its sequential placement</article-title>&#x201D; in <source>Structures of social action: Studies in conversation analysis</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>, <fpage>299</fpage>&#x2013;<lpage>345</lpage>.</citation></ref>
<ref id="ref46"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Hjalmarsson</surname> <given-names>A.</given-names></name> <name><surname>Oertel</surname> <given-names>C.</given-names></name></person-group> (<year>2012</year>). &#x201C;<article-title>Gaze direction as a back-channel inviting cue in dialogue</article-title>&#x201D; in <source>Proceedings of the IVA Workshop on &#x201C;Realtime conversational virtual agents&#x201D;</source>. vol. <volume>9</volume> (<publisher-loc>Citeseer</publisher-loc>).</citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ho</surname> <given-names>S.</given-names></name> <name><surname>Foulsham</surname> <given-names>T.</given-names></name> <name><surname>Kingstone</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Speaking and listening with the eyes: gaze signaling during dyadic interactions</article-title>. <source>PLoS One</source> <volume>10</volume>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0136905</pub-id>, PMID: <pub-id pub-id-type="pmid">26309216</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Hollenstein</surname> <given-names>T.</given-names></name></person-group> (<year>2013</year>). <source>State space grids</source>. <publisher-loc>Boston, MA</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="ref49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>Trajectories of idea emergence in dialogic collaborative problem solving: toward a complex dynamic systems perspective</article-title>. <source>Front. Psychol.</source> <volume>12</volume>:<fpage>735534</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2021.735534</pub-id>, PMID: <pub-id pub-id-type="pmid">34975626</pub-id></citation></ref>
<ref id="ref50"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Janz</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <source>Navigating common ground using feedback in conversation &#x2013; A phonetic analysis</source>. <publisher-loc>Cologne, Germany</publisher-loc>: <publisher-name>University of Cologne</publisher-name>.</citation></ref>
<ref id="ref0040"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jefferson</surname> <given-names>G.</given-names></name></person-group> (<year>1983</year>). <article-title>Two explorations of the organization of overlapping talk in conversation: Notes 915 on some orderlinesses of overlap onset</article-title>. <source>Tilburg Papers in Language and Literature</source> <volume>28</volume>, <fpage>1</fpage>&#x2013;<lpage>28</lpage>.</citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jokinen</surname> <given-names>K.</given-names></name> <name><surname>Furukawa</surname> <given-names>H.</given-names></name> <name><surname>Nishida</surname> <given-names>M.</given-names></name> <name><surname>Yamamoto</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>Gaze and turn-taking behavior in casual conversational interactions</article-title>. <source>ACM Trans. Interact. Intell. Syst.</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>30</lpage>. doi: <pub-id pub-id-type="doi">10.1145/2499474.2499481</pub-id>, PMID: <pub-id pub-id-type="pmid">39076787</pub-id></citation></ref>
<ref id="ref52"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Jokinen</surname> <given-names>K.</given-names></name> <name><surname>Nishida</surname> <given-names>M.</given-names></name> <name><surname>Yamamoto</surname> <given-names>S.</given-names></name></person-group> (<year>2009</year>). <article-title>Eye-gaze experiments for conversation monitoring</article-title>. In <conf-name>Proceedings of the 3rd International Universal Communication Symposium</conf-name> (pp. <fpage>303</fpage>&#x2013;<lpage>308</lpage>)</citation></ref>
<ref id="ref53"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Jurafsky</surname> <given-names>D.</given-names></name> <name><surname>Shriberg</surname> <given-names>E.</given-names></name> <name><surname>Fox</surname> <given-names>B.</given-names></name> <name><surname>Curl</surname> <given-names>T.</given-names></name></person-group> (<year>1998</year>). &#x201C;<article-title>Lexical, prosodic, and syntactic cues for dialog acts</article-title>,&#x201D; in <source>Proceedings of the ACL Workshop &#x201C;Discourse Relations and Discourse Markers&#x201D;</source>. <fpage>114</fpage>&#x2013;<lpage>120</lpage>.</citation></ref>
<ref id="ref54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kalma</surname> <given-names>A.</given-names></name></person-group> (<year>1992</year>). <article-title>Gazing in triads: a powerful signal in floor apportionment</article-title>. <source>Br. J. Soc. Psychol.</source> <volume>31</volume>, <fpage>21</fpage>&#x2013;<lpage>39</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2044-8309.1992.tb00953.x</pub-id></citation></ref>
<ref id="ref55"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Kato</surname> <given-names>H.</given-names></name> <name><surname>Takemoto</surname> <given-names>H.</given-names></name> <name><surname>Nishimura</surname> <given-names>R.</given-names></name> <name><surname>Mokhtari</surname> <given-names>P.</given-names></name></person-group> (<year>2010</year>). <article-title>Spatial acoustic cues for the auditory perception of speaker&#x2019;s facing direction</article-title>. <conf-name>In Proc. of 20th International Congress on Acoustics, ICA 2010. Sydney, Australia</conf-name></citation></ref>
<ref id="ref56"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Kawahara</surname> <given-names>T.</given-names></name> <name><surname>Iwatate</surname> <given-names>T.</given-names></name> <name><surname>Takanashi</surname> <given-names>K.</given-names></name></person-group> (<year>2012</year>). <article-title>Prediction of turn-taking by combining prosodic and eye-gaze information in poster conversations</article-title>. In <person-group person-group-type="editor"><name><surname>Sproat</surname> <given-names>R.</given-names></name></person-group> (Ed.), <conf-name>Proceedings of Interspeech</conf-name> (pp. <fpage>727</fpage>&#x2013;<lpage>730</lpage>).</citation></ref>
<ref id="ref57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kendon</surname> <given-names>A.</given-names></name></person-group> (<year>1967</year>). <article-title>Some functions of gaze direction in social interaction</article-title>. <source>Acta Psychol.</source> <volume>26</volume>, <fpage>22</fpage>&#x2013;<lpage>63</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0001-6918(67)90005-4</pub-id>, PMID: <pub-id pub-id-type="pmid">6043092</pub-id></citation></ref>
<ref id="ref58"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kendon</surname> <given-names>A.</given-names></name></person-group> (<year>1990</year>). <source>Conducting interaction: Patterns of behavior in focused encounters</source>, vol. <volume>7</volume>: <publisher-name>CUP Archive</publisher-name>.</citation></ref>
<ref id="ref59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kendrick</surname> <given-names>K. H.</given-names></name> <name><surname>Holler</surname> <given-names>J.</given-names></name> <name><surname>Levinson</surname> <given-names>S. C.</given-names></name></person-group> (<year>2023</year>). <article-title>Turn-taking in human face-to-face interaction is multimodal: gaze direction and manual gestures aid the coordination of turn transitions</article-title>. <source>Philos. Trans. Royal Soc B</source> <volume>378</volume>:<fpage>20210473</fpage>. doi: <pub-id pub-id-type="doi">10.1098/rstb.2021.0473</pub-id>, PMID: <pub-id pub-id-type="pmid">36871587</pub-id></citation></ref>
<ref id="ref60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kraut</surname> <given-names>R. E.</given-names></name> <name><surname>Lewis</surname> <given-names>S. H.</given-names></name> <name><surname>Swezey</surname> <given-names>L. W.</given-names></name></person-group> (<year>1982</year>). <article-title>Listener responsiveness and the coordination of conversation</article-title>. <source>J. Personal. Soc. Psychol.</source> <volume>43</volume>, <fpage>718</fpage>&#x2013;<lpage>731</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0022-3514.43.4.718</pub-id></citation></ref>
<ref id="ref61"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Lambertz</surname> <given-names>K.</given-names></name></person-group> (<year>2011</year>). <conf-name>Back-channelling: the use of yeah and mm to portray engaged listenership. Griffith Working Papers in Pragmatics and Intercultural Communication</conf-name>, <volume>4</volume>, <fpage>11</fpage>&#x2013;<lpage>18</lpage>.</citation></ref>
<ref id="ref62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lerner</surname> <given-names>G. H.</given-names></name></person-group> (<year>2003</year>). <article-title>Selecting next speaker: the context-sensitive operation of a context-free organization</article-title>. <source>Lang. Soc.</source> <volume>32</volume>, <fpage>177</fpage>&#x2013;<lpage>201</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S004740450332202X</pub-id></citation></ref>
<ref id="ref63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Levinson</surname> <given-names>S. C.</given-names></name> <name><surname>Torreira</surname> <given-names>F.</given-names></name></person-group> (<year>2015</year>). <article-title>Timing in turn-taking and its implications for processing models of language</article-title>. <source>Front. Psychol.</source> <volume>6</volume>:<fpage>731</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2015.00731</pub-id>, PMID: <pub-id pub-id-type="pmid">26124727</pub-id></citation></ref>
<ref id="ref64"><citation citation-type="book"><person-group person-group-type="author"><name><surname>L&#x00FC;cking</surname> <given-names>A.</given-names></name> <name><surname>Ptock</surname> <given-names>S.</given-names></name> <name><surname>Bergmann</surname> <given-names>K.</given-names></name></person-group> (<year>2011</year>). &#x201C;<article-title>Assessing agreement on segmentations by means of staccato, the segmentation agreement calculator according to Thomann</article-title>,&#x201D; in <source>International Gesture Workshop</source> (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer Berlin Heidelberg</publisher-name>), <fpage>129</fpage>&#x2013;<lpage>138</lpage>.</citation></ref>
<ref id="ref65"><citation citation-type="book"><person-group person-group-type="author"><name><surname>McNeill</surname> <given-names>D.</given-names></name></person-group> (<year>2005</year>). <source>Gesture, gaze, and ground. In machine learning for multimodal interaction: second international workshop</source>. <publisher-loc>Berlin Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="ref66"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Myszka</surname> <given-names>T. J.</given-names></name></person-group> (<year>1975</year>). <article-title>Situational and intrapersonal determinants of eye contact, direction of gaze aversion, smiling and other non-verbal behaviors during an interview [doctoral dissertation, University of Windsor]. Electronic Thesis and Dissertations University of Windsor</article-title>. Available online at: <ext-link xlink:href="https://scholar.uwindsor.ca/etd/3477" ext-link-type="uri">https://scholar.uwindsor.ca/etd/3477</ext-link></citation></ref>
<ref id="ref67"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Nakano</surname> <given-names>A. Y.</given-names></name> <name><surname>Yamamoto</surname> <given-names>K.</given-names></name> <name><surname>Nakagawa</surname> <given-names>S.</given-names></name></person-group> (<year>2008</year>). <article-title>Auditory perception of speaker's position, distance and facing angle in a real enclosed environment</article-title>. In <conf-name>Proc. of Autumn Meeting of Acoustic Society of Japan</conf-name> (pp. <fpage>525</fpage>&#x2013;<lpage>526</lpage>).</citation></ref>
<ref id="ref68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neiberg</surname> <given-names>D.</given-names></name> <name><surname>Gustafson</surname> <given-names>J.</given-names></name></person-group> (<year>2011</year>). <article-title>Predicting speaker changes and listener responses with and without eye-contact</article-title>. <source>Proc. Interspeech</source> <volume>1565-1568</volume>. doi: <pub-id pub-id-type="doi">10.21437/Interspeech.2011-471</pub-id></citation></ref>
<ref id="ref69"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Novick</surname> <given-names>D.</given-names></name> <name><surname>Hansen</surname> <given-names>B.</given-names></name> <name><surname>Ward</surname> <given-names>K.</given-names></name></person-group> (<year>1996</year>). <article-title>Coordinating turn-taking with gaze</article-title>. In <conf-name>Proceedings of Fourth International Conference on Spoken Language Processing (ICSLP-96)</conf-name> (pp. <fpage>1888</fpage>&#x2013;<lpage>1891</lpage>)</citation></ref>
<ref id="ref70"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Oertel</surname> <given-names>C.</given-names></name> <name><surname>W&#x0142;odarczak</surname> <given-names>M.</given-names></name> <name><surname>Edlund</surname> <given-names>J.</given-names></name> <name><surname>Wagner</surname> <given-names>P.</given-names></name> <name><surname>Gustafson</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <article-title>Gaze patterns in turn-taking</article-title>. In <person-group person-group-type="editor"><name><surname>Sproat</surname> <given-names>R.</given-names></name></person-group> (Ed.), <conf-name>Proceedings of Interspeech 2012</conf-name> (pp. <fpage>2246</fpage>&#x2013;<lpage>2249</lpage>)</citation></ref>
<ref id="ref71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ond&#x00E1;&#x0161;</surname> <given-names>S.</given-names></name> <name><surname>Kiktov&#x00E1;</surname> <given-names>E.</given-names></name> <name><surname>Pleva</surname> <given-names>M.</given-names></name> <name><surname>Juh&#x00E1;r</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Analysis of backchannel inviting cues in dyadic speech communication</article-title>. <source>Electronics</source> <volume>12</volume>:<fpage>3705</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics12173705</pub-id></citation></ref>
<ref id="ref72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paulmann</surname> <given-names>S.</given-names></name> <name><surname>Jessen</surname> <given-names>S.</given-names></name> <name><surname>Kotz</surname> <given-names>S. A.</given-names></name></person-group> (<year>2009</year>). <article-title>Investigating the multimodal nature of human communication</article-title>. <source>J. Psychophysiol.</source> <volume>23</volume>, <fpage>63</fpage>&#x2013;<lpage>76</lpage>. doi: <pub-id pub-id-type="doi">10.1027/0269-8803.23.2.63</pub-id></citation></ref>
<ref id="ref73"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Poppe</surname> <given-names>R.</given-names></name> <name><surname>Truong</surname> <given-names>K. P.</given-names></name> <name><surname>Heylen</surname> <given-names>D</given-names></name></person-group>. (<year>2011</year>). <article-title>Backchannels: quantity, type and timing matters</article-title>. In <conf-name>Intelligent Virtual Agents: 10th International Conference, IVA 2011, Reykjavik, Iceland, September 15&#x2013;17, 2011. Proceedings</conf-name> <volume>11</volume> (pp. <fpage>228</fpage>&#x2013;<lpage>239</lpage>)</citation></ref>
<ref id="ref74"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rasenberg</surname> <given-names>M.</given-names></name> <name><surname>Pouw</surname> <given-names>W.</given-names></name> <name><surname>&#x00D6;zy&#x00FC;rek</surname> <given-names>A.</given-names></name> <name><surname>Dingemanse</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>The multimodal nature of communicative efficiency in social interaction</article-title>. <source>Sci. Report</source> <volume>12</volume>:<fpage>19111</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-022-22883-w</pub-id>, PMID: <pub-id pub-id-type="pmid">36351949</pub-id></citation></ref>
<ref id="ref75"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Renklint</surname> <given-names>E.</given-names></name> <name><surname>Cardell</surname> <given-names>F.</given-names></name> <name><surname>Dahlb&#x00E4;ck</surname> <given-names>J.</given-names></name> <name><surname>Edlund</surname> <given-names>J.</given-names></name> <name><surname>Heldner</surname> <given-names>M.</given-names></name></person-group> (<year>2012</year>). <article-title>Conversational gaze in light and darkness</article-title>. <source>Proc. Fonetik</source> <volume>2012</volume>, <fpage>59</fpage>&#x2013;<lpage>60</lpage>. Available at: <ext-link xlink:href="https://www.diva-portal.org/smash/get/diva2:546623/FULLTEXT01" ext-link-type="uri">https://www.diva-portal.org/smash/get/diva2:546623/FULLTEXT01</ext-link></citation></ref>
<ref id="ref76"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Risko</surname> <given-names>E. F.</given-names></name> <name><surname>Richardson</surname> <given-names>D. C.</given-names></name> <name><surname>Kingstone</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>Breaking the fourth wall of cognitive science: real-world social attention and the dual function of gaze</article-title>. <source>Curr. Dir. Psychol. Sci.</source> <volume>25</volume>, <fpage>70</fpage>&#x2013;<lpage>74</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0963721415617806</pub-id></citation></ref>
<ref id="ref77"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Rossano</surname> <given-names>F.</given-names></name></person-group> (<year>2012</year>). <article-title>Gaze behaviour in face-to-face interaction [doctoral dissertation, Radboud university, Nijmegen]</article-title>. Available online at: <ext-link xlink:href="https://hdl.handle.net/11858/00-001M-0000-000F-ED23-5" ext-link-type="uri">https://hdl.handle.net/11858/00-001M-0000-000F-ED23-5</ext-link></citation></ref>
<ref id="ref78"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rossano</surname> <given-names>F.</given-names></name> <name><surname>Brown</surname> <given-names>P.</given-names></name> <name><surname>Levinson</surname> <given-names>S. C.</given-names></name></person-group> (<year>2009</year>). <article-title>Gaze, questioning and culture</article-title>. <source>Convers. Analys.</source> <volume>27</volume>, <fpage>187</fpage>&#x2013;<lpage>249</lpage>. doi: <pub-id pub-id-type="doi">10.1017/CBO9780511635670.008</pub-id>, PMID: <pub-id pub-id-type="pmid">40166671</pub-id></citation></ref>
<ref id="ref79"><citation citation-type="book"><person-group person-group-type="author"><name><surname>R&#x00FC;hlemann</surname> <given-names>C.</given-names></name></person-group> (<year>2007</year>). <source>Conversation in context: A corpus-driven approach</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Longman</publisher-name>.</citation></ref>
<ref id="ref80"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rutter</surname> <given-names>D. R.</given-names></name> <name><surname>Stephenson</surname> <given-names>G. M.</given-names></name> <name><surname>Ayling</surname> <given-names>K.</given-names></name> <name><surname>White</surname> <given-names>P. A.</given-names></name></person-group> (<year>1978</year>). <article-title>The timing of looks in dyadic conversation</article-title>. <source>British J. Soc. Clinic. Psychol.</source> <volume>17</volume>, <fpage>17</fpage>&#x2013;<lpage>21</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2044-8260.1978.tb00890.x</pub-id></citation></ref>
<ref id="ref81"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sacks</surname> <given-names>H.</given-names></name> <name><surname>Schegloff</surname> <given-names>E. A.</given-names></name> <name><surname>Jefferson</surname> <given-names>G.</given-names></name></person-group> (<year>1974</year>). <article-title>A simplest systematics for the organization of turn-taking for conversation</article-title>. <source>Language</source> <volume>50</volume>, <fpage>696</fpage>&#x2013;<lpage>735</lpage>. doi: <pub-id pub-id-type="doi">10.2307/412243</pub-id></citation></ref>
<ref id="ref82"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Savino</surname> <given-names>M.</given-names></name></person-group> (<year>2010</year>). &#x201C;<article-title>Intonational strategies for backchanneling in Italian Map Task dialogues</article-title>,&#x201D; in <source>Proceedings of the 3rd ISCA Workshop ExLing 2010</source>.</citation></ref>
<ref id="ref83"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Savino</surname> <given-names>M.</given-names></name></person-group> (<year>2011</year>). <conf-name>The intonation of backchannels in Italian task-oriented dialogues: Cues to turn-taking dynamics, information status and speaker&#x2019;s attitude. In Proceedings of the 5th Language and Technology Conference: Human Language Technology as a Challenge for Computer Science and Linguistics</conf-name>.</citation></ref>
<ref id="ref84"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Savino</surname> <given-names>M.</given-names></name></person-group> (<year>2012</year>). <article-title>The intonation of polar questions in Italian: Where is the rise?</article-title> <source>J. Int. Phonetic Assoc.</source> <volume>42</volume>, <fpage>23</fpage>&#x2013;<lpage>48</lpage>.</citation></ref>
<ref id="ref85"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Savino</surname> <given-names>M.</given-names></name> <name><surname>Lapertosa</surname> <given-names>L.</given-names></name> <name><surname>Refice</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Seeing or Not Seeing Your Conversational Partner: The Influence of Interaction Modality on Prosodic Entrainment</article-title>&#x201D; in <source>Speech and Computer</source>. eds. <person-group person-group-type="editor"><name><surname>Karpov</surname> <given-names>A.</given-names></name> <name><surname>Jokisch</surname> <given-names>O.</given-names></name> <name><surname>Potapova</surname> <given-names>R.</given-names></name></person-group> (<publisher-loc>Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation></ref>
<ref id="ref86"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Savino</surname> <given-names>M.</given-names></name> <name><surname>Refice</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). <conf-name>Acknowledgement or reply? Prosodic features for disambiguating pragmatic functions of the Italian token &#x201C;s&#x00EC;.&#x201D; In Proceedings of the 7th Conference on Speech Technology and Human-Computer Dialogue (SpeD)</conf-name> (pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>).</citation></ref>
<ref id="ref87"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Sbranna</surname> <given-names>S.</given-names></name> <name><surname>Wehrle</surname> <given-names>S.</given-names></name> <name><surname>Grice</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). &#x201C;<article-title>The use of Backchannels and other Very Short Utterances by Italian Learners of German</article-title>&#x201D; in <source>La posizione del parlante nell'interazione: atteggiamenti, intenzioni ed emozioni nella comunicazione verbale [The position of the speaker in interaction: attitudes, intentions, and emotions in verbal communication]</source>. eds. <person-group person-group-type="editor"><name><surname>Orrico</surname> <given-names>R.</given-names></name> <name><surname>Schettino</surname> <given-names>L.</given-names></name></person-group> (<publisher-loc>Milan</publisher-loc>: <publisher-name>Officinaventuno</publisher-name>).</citation></ref>
<ref id="ref88"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sbranna</surname> <given-names>S.</given-names></name> <name><surname>Wehrle</surname> <given-names>S.</given-names></name> <name><surname>Grice</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>A multi-dimensional analysis of backchannels in L1 German, L1 Italian and L2 German</article-title>. <source>Lang. Interact. Acquis.</source> <volume>15.2</volume>, <fpage>242</fpage>&#x2013;<lpage>276</lpage>. doi: <pub-id pub-id-type="doi">10.1075/lia.00026.sbr</pub-id></citation></ref>
<ref id="ref89"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Schegloff</surname> <given-names>E. A.</given-names></name></person-group> (<year>1982</year>). &#x201C;<article-title>Discourse as an interactional achievement: Some uses of &#x2018;uh huh&#x2019; and other things that come between sentences</article-title>&#x201D; in <source>Analysing discourse: Text and talk</source>. ed. <person-group person-group-type="editor"><name><surname>Tannen</surname> <given-names>D.</given-names></name></person-group> (<publisher-loc>Washington, D.C.</publisher-loc>: <publisher-name>Georgetown University Press</publisher-name>), <fpage>71</fpage>&#x2013;<lpage>93</lpage>.</citation></ref>
<ref id="ref90"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shelley</surname> <given-names>L.</given-names></name> <name><surname>Gonzalez</surname> <given-names>F.</given-names></name></person-group> (<year>2013</year>). <article-title>Back channeling: function of back channeling and L1 effects on back channeling in L2</article-title>. <source>Linguis. Portfolios</source> <volume>2</volume>:<fpage>9</fpage>. Available at: <ext-link xlink:href="https://repository.stcloudstate.edu/stcloud_ling/vol2/iss1/9" ext-link-type="uri">https://repository.stcloudstate.edu/stcloud_ling/vol2/iss1/9</ext-link></citation></ref>
<ref id="ref91"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simon</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>The functions of active listening responses</article-title>. <source>Behav. Process.</source> <volume>157</volume>, <fpage>47</fpage>&#x2013;<lpage>53</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.beproc.2018.08.013</pub-id>, PMID: <pub-id pub-id-type="pmid">30195899</pub-id></citation></ref>
<ref id="ref92"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Spaniol</surname> <given-names>M.</given-names></name> <name><surname>Janz</surname> <given-names>A.</given-names></name> <name><surname>Wehrle</surname> <given-names>S.</given-names></name> <name><surname>Vogeley</surname> <given-names>K.</given-names></name> <name><surname>Grice</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). &#x201C;<article-title>Multimodal signalling: the interplay of oral and visual feedback in conversation</article-title>,&#x201D; in <source>Proceedings of the 20th international congress of phonetic sciences</source>.</citation></ref>
<ref id="ref93"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Stahl</surname> <given-names>G.</given-names></name></person-group> (<year>2016</year>). <source>Constructing dynamic triangles together: The development of mathematical group cognition</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation></ref>
<ref id="ref94"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Streeck</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Mutual gaze and recognition. Revisiting Kendon&#x2019;s gaze direction in two-person conversation</article-title>&#x201D; in <source>From gesture in conversation to visible action as utterance</source>. eds. <person-group person-group-type="editor"><name><surname>Seyfeddinipur</surname> <given-names>M.</given-names></name> <name><surname>Gullberg</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>John Benjamins</publisher-name>), <fpage>35</fpage>&#x2013;<lpage>55</lpage>.</citation></ref>
<ref id="ref95"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll1">The Language Archive</collab></person-group>. (<year>2023</year>). <article-title>Nijmegen: Max Planck Institute for Psycholinguistics, ELAN (Version 6.7)</article-title>. Available online at: <ext-link xlink:href="https://archive.mpi.nl/tla/elan" ext-link-type="uri">https://archive.mpi.nl/tla/elan</ext-link></citation></ref>
<ref id="ref96"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Truong</surname> <given-names>K. P.</given-names></name> <name><surname>Poppe</surname> <given-names>R.</given-names></name> <name><surname>Kok</surname> <given-names>I. D.</given-names></name> <name><surname>Heylen</surname> <given-names>D.</given-names></name></person-group> (<year>2011</year>). <article-title>A multimodal analysis of vocal and visual backchannels in spontaneous dialogs</article-title>. <source>Proc. Interspeech</source> <volume>2011</volume>, <fpage>2973</fpage>&#x2013;<lpage>2976</lpage>. doi: <pub-id pub-id-type="doi">10.21437/Interspeech.2011-744</pub-id></citation></ref>
<ref id="ref97"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Wehrle</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <source>Conversation and intonation in autism: A multi-dimensional analysis</source>. <publisher-loc>Berlin</publisher-loc>: <publisher-name>Language Science Press</publisher-name>.</citation></ref>
<ref id="ref98"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Wittenburg</surname> <given-names>P.</given-names></name> <name><surname>Brugman</surname> <given-names>H.</given-names></name> <name><surname>Russel</surname> <given-names>A.</given-names></name> <name><surname>Klassmann</surname> <given-names>A.</given-names></name> <name><surname>Sloetjes</surname> <given-names>H</given-names></name></person-group>. (<year>2006</year>). <source>ELAN: A professional framework for multimodality research</source>. In <conf-name>Proceedings of LREC 2006, Fifth International Conference on Language Resources and Evaluation</conf-name>.</citation></ref>
<ref id="ref99"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yngve</surname> <given-names>V. H.</given-names></name></person-group> (<year>1970</year>). <article-title>On getting a word in edgewise</article-title>. <source>Chicago Linguistics Society</source> <volume>6</volume>, <fpage>567</fpage>&#x2013;<lpage>578</lpage>.</citation></ref>
<ref id="ref100"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zima</surname> <given-names>E.</given-names></name> <name><surname>Wei&#x00DF;</surname> <given-names>C.</given-names></name> <name><surname>Br&#x00F4;ne</surname> <given-names>G.</given-names></name></person-group> (<year>2019</year>). <article-title>Gaze and overlap resolution in triadic interactions</article-title>. <source>J. Pragmat.</source> <volume>140</volume>, <fpage>49</fpage>&#x2013;<lpage>69</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.pragma.2018.11.019</pub-id></citation></ref>
</ref-list>
</back>
</article>