<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1479905</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Dense Paraphrasing for multimodal dialogue interpretation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Tu</surname> <given-names>Jingxuan</given-names></name>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/821561/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Rim</surname> <given-names>Kyeongmin</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2906473/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ye</surname> <given-names>Bingyang</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2905924/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lai</surname> <given-names>Kenneth</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2906930/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Pustejovsky</surname> <given-names>James</given-names></name>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/859904/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff><institution>Computer Science Department, Brandeis University</institution>, <addr-line>Waltham, MA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Raymond Lee, Beijing Normal University-Hong Kong Baptist University United International College, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Seon Phil Jeong, United International College, China</p>
<p>Zhiyuan Li, United International College, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Jingxuan Tu <email>jxtu&#x00040;brandeis.edu</email></corresp>
<corresp id="c002">James Pustejovsky <email>jamesp&#x00040;brandeis.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>12</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1479905</elocation-id>
<history>
<date date-type="received">
<day>13</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Tu, Rim, Ye, Lai and Pustejovsky.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Tu, Rim, Ye, Lai and Pustejovsky</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Multimodal dialogue involving multiple participants presents complex computational challenges, primarily due to the rich interplay of diverse communicative modalities including speech, gesture, action, and gaze. These modalities interact in complex ways that traditional dialogue systems often struggle to accurately track and interpret. To address these challenges, we extend the textual enrichment strategy of Dense Paraphrasing (DP), by translating each nonverbal modality into linguistic expressions. By normalizing multimodal information into a language-based form, we hope to both simplify the representation for and enhance the computational understanding of situated dialogues. We show the effectiveness of the dense paraphrased language form by evaluating instruction-tuned Large Language Models (LLMs) against the Common Ground Tracking (CGT) problem using a publicly available collaborative problem-solving dialogue dataset. Instead of using multimodal LLMs, the dense paraphrasing technique represents the dialogue information from multiple modalities in a compact and structured machine-readable text format that can be directly processed by the language-only models. We leverage the capability of LLMs to transform machine-readable paraphrases into human-readable paraphrases, and show that this process can further improve the result on the CGT task. Overall, the results show that augmenting the context with dense paraphrasing effectively facilitates the LLMs&#x00027; alignment of information from multiple modalities, and in turn largely improves the performance of common ground reasoning over the baselines. Our proposed pipeline with original utterances as input context already achieves comparable results to the baseline that utilized decontextualized utterances which contain rich coreference information. When also using the decontextualized input, our pipeline largely improves the performance of common ground reasoning over the baselines. We discuss the potential of DP to create a robust model that can effectively interpret and integrate the subtleties of multimodal communication, thereby improving dialogue system performance in real-world settings.</p></abstract>
<kwd-group>
<kwd>Dense Paraphrasing</kwd>
<kwd>Common Ground Tracking</kwd>
<kwd>dialogue system</kwd>
<kwd>Large Language Models</kwd>
<kwd>multimodal communication</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Science Foundation<named-content content-type="fundref-id">10.13039/100000001</named-content></contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="7"/>
<equation-count count="0"/>
<ref-count count="90"/>
<page-count count="15"/>
<word-count count="11977"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Natural Language Processing</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Modeling the interpretation of multimodal dialogue remains a challenging task, both formally and computationally (Saha et al., <xref ref-type="bibr" rid="B67">2018</xref>; Liao et al., <xref ref-type="bibr" rid="B48">2018</xref>). It involves not only aligning and composing the meanings conveyed through the different modalities, such as speech, gesture, and gaze, but also identifying actions and contextual factors occuring during the interaction. Traditionally, dialogue systems have had difficulty tracking and interpreting the diverse interactions between multiple communicative modalities, particularly when faced with the problem of underspecified references (Vinyals and Le, <xref ref-type="bibr" rid="B81">2015</xref>; Baltru&#x00161;aitis et al., <xref ref-type="bibr" rid="B4">2018</xref>).</p>
<p>When engaged in dialogue, our shared understanding of both utterance meaning (content) and the speaker&#x00027;s meaning in a specific context (intent) involves the ability to link these two in the act of situationally grounding meaning to the local context&#x02014;what is typically referred to as &#x0201C;establishing the common ground&#x0201D; between speakers (Clark and Brennan, <xref ref-type="bibr" rid="B19">1991</xref>; Traum, <xref ref-type="bibr" rid="B75">1994</xref>; Asher and Gillies, <xref ref-type="bibr" rid="B3">2003</xref>; Dillenbourg and Traum, <xref ref-type="bibr" rid="B26">2006</xref>). The concept of common ground refers to the set of shared beliefs among participants in Human-Human interaction (HHI) (Traum, <xref ref-type="bibr" rid="B75">1994</xref>; Hadley et al., <xref ref-type="bibr" rid="B38">2022</xref>), as well as Human-Computer Interaction (HCI) (Krishnaswamy and Pustejovsky, <xref ref-type="bibr" rid="B44">2019</xref>; Ohmer et al., <xref ref-type="bibr" rid="B57">2022</xref>) and Human-Robot Interaction (HRI) (Kruijff et al., <xref ref-type="bibr" rid="B45">2010</xref>; Fischer, <xref ref-type="bibr" rid="B33">2011</xref>; Scheutz et al., <xref ref-type="bibr" rid="B68">2011</xref>). Researchers have recently employed the notion of common ground operationally to identify and select relevant information for conversational Question Answering (QA) system design (Nishida, <xref ref-type="bibr" rid="B55">2018</xref>; Del Tredici et al., <xref ref-type="bibr" rid="B22">2022</xref>).</p>
<p>In conversational multimodal dialogue systems, it is not enough to simply recognize individual modalities, such as speech, gesture, or gaze, in isolation. The true challenge lies in the accurate alignment and integration of these modalities to derive a cohesive understanding of the dialogue context. For instance, the subtle yet critical co-attention between participants&#x02014;where both parties focus on the same object or region of interest&#x02014;can dramatically shift the meaning of an utterance. If a system fails to detect or properly integrate these multimodal cues, the resulting interpretation may be incomplete or even incorrect, leading to misunderstandings and breakdowns in communication.</p>
<p>Underspecified references, such as pronouns and demonstratives, are frequently used in natural conversation to refer to entities that are contextually salient but not explicitly named. This reliance on shared context can lead to ambiguities that are challenging for dialogue systems to resolve (Byron, <xref ref-type="bibr" rid="B13">2002</xref>; Eckert and Strube, <xref ref-type="bibr" rid="B28">2000</xref>; M&#x000FC;ller, <xref ref-type="bibr" rid="B53">2008</xref>; Khosla et al., <xref ref-type="bibr" rid="B43">2021</xref>).</p>
<p>For example, when a speaker says &#x0201C;one of <italic>those</italic>&#x0201D; while pointing at an object, as in <xref ref-type="fig" rid="F1">Figure 1</xref>, the word itself is insufficient to convey the full meaning without considering the accompanying gesture. The integration of visual cues from gestures and gaze with linguistic information allows the system to disambiguate these references by narrowing down the possible entities being referred to. Moreover, the synchronization of gestures with speech provides additional semantic information, such as emphasis or referential clarification (e.g., the locational demonstrative <italic>there</italic> in <xref ref-type="fig" rid="F1">Figure 1</xref>), that is crucial for understanding the speaker&#x00027;s intent.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Example of a triad (three participants, P1, P2, P3) multimodal interaction in the weights task: P1 (left) says: &#x0201C;Put one of those on there.&#x0201D;; purple box denotes P1 pointing to the blocks and scale; red arrows denote co-gazing by P1&#x02013;P3; blue arrows symbolize P1&#x02013;P3 leaning toward the table.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1479905-g0001.tif"/>
</fig>
<p>Consequently, the need for more robust methods to handle these ambiguities is of great importance. Advanced Artificial Intelligence (AI) systems must incorporate sophisticated multimodal fusion techniques that not only recognize each modality but also align and integrate them to form a unified representation of the dialogue context. This process involves leveraging models that can map gestures to referential expressions, correlate gaze patterns with attentional focus, and link these nonverbal cues with the linguistic content of the conversation.</p>
<p>To address this challenge, our research adopts the data augmentation technique of Dense Paraphrasing (DP) (Tu et al., <xref ref-type="bibr" rid="B78">2023</xref>; Rim et al., <xref ref-type="bibr" rid="B66">2023</xref>) to the task of interpreting multimodal dialogue. In this extension, we propose Multi-Modal Dense Paraphrasing (MMDP) that involves translating nonverbal modalities into linguistic expressions, thereby recontextualizing and clarifying the meaning of underspecified references. By creating cross-modal coreference links and binding these references with action or gesture annotations, we aim to enrich the textual content and enhance the computational understanding of dialogues.</p>
<p>We explore the utility of MMDP on the Common Ground Tracking (CGT) problem (Khebour et al., <xref ref-type="bibr" rid="B42">2024</xref>) on the recent published Weights Task Dataset (WTD) (Khebour et al., <xref ref-type="bibr" rid="B41">2023</xref>). This dataset contains videos in which groups of three were asked to determine the weights of five blocks using a balance scale. This collection contains annotations from multiple modalities recorded in the videos, as well as identification of the group epistemic state at each dialogue state. The CGT problem defined over the dataset is to identify the common ground (knowledge of the weights of different blocks) among the participants of each group. In our previous joint work (Khebour et al., <xref ref-type="bibr" rid="B42">2024</xref>), a hybrid method of neural networks and heuristics was adopted to solve the CGT problem.</p>
<p>In this paper, we instead treat CGT as a QA task that involves two steps: applying MMDP to convert information from multiple modalities into meaningful paraphrases, and then using the paraphrases as the context for prompts that ask about the common ground. We leverage Large Language Models (LLMs) for the whole pipeline and evaluate the results under different settings. We find that the human readable paraphrase generated by MMDP can better integrate the information from the dialogue context and multiple modalities, thus improving the performance over baselines by a large margin. We also compare the results by varying different models and the length of input context, providing further insights for future work. We make our source code and data publicly available.<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref></p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>Recent years have seen remarkable progress on tasks involving multimodality (Chhabra and Vishwakarma, <xref ref-type="bibr" rid="B16">2023</xref>; Das and Singh, <xref ref-type="bibr" rid="B21">2023</xref>; Zhao et al., <xref ref-type="bibr" rid="B89">2023</xref>; Gong et al., <xref ref-type="bibr" rid="B36">2023</xref>). Encoding multimodal information into embeddings involves combining data from different modalities, such as text, images, and audio, into a unified representation, and is a vital component of many multimodal tasks.</p>
<p>In recent studies, multimodal encoders are usually built upon different vector extraction algorithms for different modalities, and then a combination operation is performed over those vectors. For example, to combine language and vision modalities, Chuang et al. (<xref ref-type="bibr" rid="B18">2020</xref>) use contextualized word embeddings for language and acoustic feature extraction for audio, and then uses vector addition of the two to train an RNN model. Similarly, Sur&#x000ED;s et al. (<xref ref-type="bibr" rid="B72">2018</xref>) leverage two separate video and audio features to train shared weights. On the other hand, Khebour et al. (<xref ref-type="bibr" rid="B42">2024</xref>) use concatenation of word embeddings and more transparent k-hot encodings to encode multimodal information. More recently, multimodal LLMs such as GPT-4V (OpenAI et al., <xref ref-type="bibr" rid="B58">2024</xref>) have been used to incorporate image inputs into LLMs. Contrary to previous studies, in this paper, we leverage LLMs to map non-verbal modality data into a natural language form and then treat the text as augmented multimodal data.</p>
<p>QA is a significant area in NLP and various other NLP tasks such as Summarization (Eyal et al., <xref ref-type="bibr" rid="B32">2019</xref>; Deutsch et al., <xref ref-type="bibr" rid="B23">2021</xref>; Gunasekara et al., <xref ref-type="bibr" rid="B37">2021</xref>), Data Augmentation (Mekala et al., <xref ref-type="bibr" rid="B52">2022</xref>), and Question Generation (Tu et al., <xref ref-type="bibr" rid="B79">2022b</xref>) can be enhanced by integrating QA techniques. We leverage QA to facilitate the tracking of common ground in situated dialogue in this work. The general goal of Dialogue State Tracking (DST) is to maintain and update the state of dialogue by accurately tracking user intents and belief states during a multi-turn conversation (Budzianowski et al., <xref ref-type="bibr" rid="B11">2018</xref>; Liao et al., <xref ref-type="bibr" rid="B47">2021</xref>; Jacqmin et al., <xref ref-type="bibr" rid="B39">2022</xref>). Del Tredici et al. (<xref ref-type="bibr" rid="B22">2022</xref>) introduce CGT as a mitigation method for conversational QA. The task aims to estimate the shared understanding or &#x0201C;common ground&#x0201D; between the conversational participants. DST focuses on task completion within a single session and deals with specific slots and intents related to the task, while CGT focuses on maintaining mutual understanding throughout the conversation with broader shared knowledge and assumptions. Khebour et al. (<xref ref-type="bibr" rid="B42">2024</xref>) is the first attempt to apply CGT over real-world multiparty dialogue instead of just conversational QA.</p>
<p>Textual enrichment has been employed to address the challenge of understanding the economy of sentence structure in comprehension tasks. Approaches to textual enrichment include paraphrasing (Bhagat and Hovy, <xref ref-type="bibr" rid="B7">2013</xref>; Barzilay and Elhadad, <xref ref-type="bibr" rid="B5">1997</xref>) and decontextualization (Choi et al., <xref ref-type="bibr" rid="B17">2021</xref>; Elazar et al., <xref ref-type="bibr" rid="B30">2021</xref>; Wu et al., <xref ref-type="bibr" rid="B82">2021</xref>). DP has been recently introduced in Tu et al. (<xref ref-type="bibr" rid="B78">2023</xref>) as a linguistically motivated textual enrichment strategy and has been leveraged to facilitate a variety of NLP tasks such as Coreference Resolution (Rim et al., <xref ref-type="bibr" rid="B66">2023</xref>), Completion (Ye B. et al., <xref ref-type="bibr" rid="B83">2022</xref>), and Meaning Representation (Tu et al., <xref ref-type="bibr" rid="B77">2024</xref>). Khebour et al. (<xref ref-type="bibr" rid="B42">2024</xref>) also used DP to recover propositional content (and subsequent sentence embeddings) in user utterances in multimodal data. We further extend the usage of DP to translate nonverbal modalities into linguistic expressions in the broader context of Natural Language Generation (NLG), the task of generating natural language text from a knowledge base or logical form representation. NLG is a crucial component of QA and dialogue systems. Traditional NLG methods are mostly rule-based (Bateman and Henschel, <xref ref-type="bibr" rid="B6">1999</xref>; Busemann and Horacek, <xref ref-type="bibr" rid="B12">1998</xref>), while later works approach the problem with neural networks (Zhou et al., <xref ref-type="bibr" rid="B90">2016</xref>; Tran and Nguyen, <xref ref-type="bibr" rid="B74">2018</xref>). With the recent advances in LLMs, such models (Touvron et al., <xref ref-type="bibr" rid="B73">2023</xref>; Achiam et al., <xref ref-type="bibr" rid="B1">2023</xref>) show great capabilities in generation tasks. In this paper, we leverage LLMs to facilitate DP and generate answers for CGT questions.</p>
</sec>
<sec id="s3">
<title>3 Theory and practice of dense paraphrasing</title>
<p>In this section, we introduce the textual enrichment and data augmentation strategy of Dense Paraphrasing (DP), and describe how it enables deeper capabilities in computational Natural Language Understanding (NLU) models.</p>
<sec>
<title>3.1 Background and definition</title>
<p>NLU has long been considered a fundamental task within AI, involving both parsing and understanding the semantics of language inputs, including grammar, context, and intent. Such work has focused on enabling machines to perform tasks like sentiment analysis, question answering, information extraction, and information retrieval effectively.</p>
<p>NLU, however, remains an extremely difficult task, particularly when deployed in the service of dialogue understanding and conversation analysis (Ye F. et al., <xref ref-type="bibr" rid="B84">2022</xref>; Yi et al., <xref ref-type="bibr" rid="B85">2024</xref>; Ou et al., <xref ref-type="bibr" rid="B60">2024</xref>).</p>
<p>Furthermore, despite the fast-paced growth of AI, advanced computational models are still challenged by natural language partly due to lacking a deeper understanding of the economy of sentence structures. We, as humans, interpret sentences as contextualized components of a narrative or discourse, by both filling in missing information, and reasoning about event consequences. However, most existing language models understand inferences from text merely by recovering surface arguments, adjuncts, or strings associated with the query terms or prompts (Parikh et al., <xref ref-type="bibr" rid="B62">2016</xref>; Chen et al., <xref ref-type="bibr" rid="B15">2017</xref>; Kumar and Talukdar, <xref ref-type="bibr" rid="B46">2020</xref>; Schick and Sch&#x000FC;tze, <xref ref-type="bibr" rid="B69">2021</xref>).</p>
<p>Prior work on improving NLU systems to learn beyond the surface texts has taken two directions. The first involves commonsense reasoning and knowledge understanding (Poria et al., <xref ref-type="bibr" rid="B63">2014</xref>; Angeli and Manning, <xref ref-type="bibr" rid="B2">2014</xref>; Emami et al., <xref ref-type="bibr" rid="B31">2018</xref>; Mao et al., <xref ref-type="bibr" rid="B50">2019</xref>; Lin et al., <xref ref-type="bibr" rid="B49">2021</xref>), both of which improve NLU models by providing the ability to make inferences and interpret nuances from knowledge about the everyday world, and concepts of entities from knowledge bases.</p>
<p>The second line of work involves data augmentation over the input. This approach focuses on paraphrasing or enriching the texts by increasing the variability in the text format, and reducing the dependency on the contexts from other texts (Culicover, <xref ref-type="bibr" rid="B20">1968</xref>; Goldman, <xref ref-type="bibr" rid="B35">1977</xref>; Muraki, <xref ref-type="bibr" rid="B54">1982</xref>; Boyer and Lapalme, <xref ref-type="bibr" rid="B8">1985</xref>; McKeown, <xref ref-type="bibr" rid="B51">1983</xref>; Barzilay and Elhadad, <xref ref-type="bibr" rid="B5">1997</xref>; Bhagat and Hovy, <xref ref-type="bibr" rid="B7">2013</xref>; Choi et al., <xref ref-type="bibr" rid="B17">2021</xref>; Elazar et al., <xref ref-type="bibr" rid="B30">2021</xref>; Chai et al., <xref ref-type="bibr" rid="B14">2022</xref>; Eisenstein et al., <xref ref-type="bibr" rid="B29">2022</xref>; Tu et al., <xref ref-type="bibr" rid="B79">2022b</xref>; Ye B. et al., <xref ref-type="bibr" rid="B83">2022</xref>; Katz et al., <xref ref-type="bibr" rid="B40">2022</xref>). We argue here that such augmented texts can in turn help NLU systems to better handle the ambiguities and variants in human language, particularly when used in multimodal settings. We extend the technique of Dense Paraphrasing (DP) (Tu et al., <xref ref-type="bibr" rid="B78">2023</xref>) to multimodal interactions. DP is a technique that rewrites a textual expression to reduce ambiguity while making explicit the underlying semantics of the expression. DP reveals a set of paraphrases that act as the signature for a semantic type, which is consistent with canonical syntactic forms for a semantic type (Pustejovsky, <xref ref-type="bibr" rid="B64">1995</xref>). Here we define DP as follows:</p>
<p><bold> Definition 1</bold>. <bold>Dense Paraphrasing</bold> <bold>(DP)</bold>: Given a pair (<italic>S, P</italic>) of two expressions in a language, <italic>P</italic> is a valid <italic>Dense Paraphrase</italic> of <italic>S</italic> if <italic>P</italic> is an expression (lexeme, phrase, sentence) that, (1) <bold>[consistency]</bold> eliminates any contextual ambiguity that may be present in <italic>S</italic>; (2) <bold>[informativeness]</bold> makes explicit any underlying semantics (hidden arguments, dropped objects or adjuncts) that is not otherwise expressed in the economy of sentence structure.</p>
</sec>
<sec>
<title>3.2 Subtasks of Dense Paraphrasing</title>
<p>In practice, to achieve the said level of context-independence and generate fully self-sustained textual expressions, we include (but are not limited to) the following subtasks as the fundamental building blocks of DP augmentation:</p>
<p><bold>Anaphora and coreference</bold>: Understanding the contextual semantics of referring expressions is a crucial step for NLU. To that end, being able to dereference and then to canonicalize pronouns and other noun phrases is an integral step toward DP.</p>
<p><bold>Frame saturation</bold>: Argument structure in event semantics can provide a rich understanding of relations among event participants and causal relations between entity states (as a result of the event). However, due to the economy of natural language, the full argument structure of an event is seldom present in linguistic surface forms. Hence recovering those omitted arguments and saturating the event frames (argument structures) is another critical goal for DP.</p>
<p><bold>Event decomposition</bold>: Some events can be decomposed into multiple steps or subevents. Humans can easily understand underlying subevent structures (individual subevents and their temporal order) based on their lexical competence, and hence can use abstract vocabulary for complex actions and events in natural language. Surfacing the underlying subevent structure is another aspect of what DP aims to achieve in terms of data augmentation for NLU systems.</p>
<p><bold>Entity state tracking</bold>: Actions have consequences. Events make changes to paricipant entities and re-configure the world status. However, for the same economic reason, we humans heavily rely on prior (commonsense or empirical) knowledge to carry complex causal and temporal relations between entities through chains of events. Thus, within DP, we aim to provide temporally ordered state changes as a part of the textual enrichment strategy.</p>
<p><bold>Multimodal alignment</bold>: Motivated by the concept of DP that is first outlined in and adopted by the above work to create rich paraphrases of implicit entities represented in structured graphs, we extend DP to encode the multimodal input into a <italic>machine readable</italic> format, and then decode it into <italic>human readable</italic> paraphrases. Text in machine readable format is a form of (semi-)structured textual representation of the multimodality that is flexible enough to be ingested by the model and transformed into other formats. Text in human readable format is natural language that is more effectively processed and interpreted by language models. More implementational details are described in Section 6.2.4.</p>
</sec>
<sec>
<title>3.3 Applications of DP</title>
<p>In previous work, we proposed the textual enrichment strategy called Dense Paraphrasing (DP), and explored how it enables deeper NLU capability for computational models. DP transforms and enriches the texts that will be input to the computational models. It reflects and facilitates the models&#x00027; capability to understand the meaning of language in a way that improves downstream NLU tasks. DP differs from previous work in that it is more linguistically motivated and focuses on the realization of compositional operations inherent in the meaning of the language. This makes DP-enriched texts independent of external knowledge, relying solely on the contextualized or grounded information from the sentence or document structure.</p>
<p>The proposed DP technique helps address practical NLU tasks by providing tools, datasets, and resources that allow models to learn text more efficiently and easily by augmenting the context with traceable states for all mentions and events involved in the text. Given the context, DP can enrich the text by enriching the events with their implicit state information and linking the enriched events until the goal is reached.</p>
<p>DP has been applied to improve the logical metonymy task by surfacing implicit types through the semantic reconstruction of the sentence (Ye B. et al., <xref ref-type="bibr" rid="B83">2022</xref>). Metonymy identifies implicit meaning, such as the understood activity of &#x0201C;drinking&#x0201D; in <italic>Jon enjoyed his coffee</italic>. The paraphrased sentences with an explicated event-argument structure are used to train masked language models for the logical metonymy task.</p>
<p>Tu et al. (<xref ref-type="bibr" rid="B76">2022a</xref>,<xref ref-type="bibr" rid="B79">b</xref>) defined a QA task that applies DP to generate questions over implicit arguments and event states from procedural texts, which provided a lens into a model&#x00027;s reasoning capability in the task. The QA task includes competence-based questions that focus on queries over lexical semantic knowledge involving implicit argument and subevent structures of verbs. The paper found that the corresponding QA task is challenging for large pre-trained language models until they are provided with additional contextualized semantic information. Obiso et al. (<xref ref-type="bibr" rid="B56">2024</xref>) also demonstrated that QA tasks using DP-enriched contexts leads to increased performance on various models.</p>
<p>The DP technique has been further applied to a more challenging coreference and anaphora resolution task that involves implicit and transformed objects. Tu et al. (<xref ref-type="bibr" rid="B78">2023</xref>) applied DP on procedural texts to generate hidden arguments and explicate the transformation of the arguments from a chain of events on the surface texts. Following this, Rim et al. (<xref ref-type="bibr" rid="B66">2023</xref>) utilized the proposed event semantics for the entity transformation to represent recipe texts as I/O process graph structures that are able to better model entity coreference.</p>
<p>DP can also be used for constructing novel linguistic resources. Tu et al. (<xref ref-type="bibr" rid="B77">2024</xref>) proposed to enrich Abstract Meaning Representation (AMR) with GL-VerbNet. The paper developed a new syntax, concepts, and roles for subevent structure based on VerbNet for connecting subevents to atomic predicates. They demonstrated the application of the new AMR dataset for generating enriched paraphrases with details of subevent transformations and arguments that are not present in the surface form of the texts.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Common Ground Tracking</title>
<p>Common Ground Tracking (CGT) is the task of identifying the shared belief space held by all participants in a task-oriented dialogue (Khebour et al., <xref ref-type="bibr" rid="B42">2024</xref>). This involves finding the propositions that are acknowledged and accepted by all participants engaged in the task. In this context, we model the dialogue as a set of beliefs and the evidence supporting those beliefs at each conversational turn. Each turn may introduce, reinforce, or change beliefs, and the CGT task focuses on tracking these shared understandings throughout the dialogue. To do this, we use a Common Ground Structure (CGS), inspired by the notion of a dialogue gameboard (Ginzburg, <xref ref-type="bibr" rid="B34">2012</xref>), as well as by evidence-based dynamic epistemic logic (van Benthem et al., <xref ref-type="bibr" rid="B80">2014</xref>; Pacuit, <xref ref-type="bibr" rid="B61">2017</xref>). A CGS has three components (Example usage in Section 5.1):</p>
<list list-type="order">
<list-item><p>QB<sc>ank</sc>: set of propositions that could be true; i.e., that have not yet been ruled out;</p></list-item>
<list-item><p>EB<sc>ank</sc>: set of propositions for which there is some evidence they are true;</p></list-item>
<list-item><p>FB<sc>ank</sc>: set of propositions believed as true by the group.</p></list-item>
</list>
<p>To evaluate systems designed for CGT, we formulate it as a QA task. In this setup, the system is prompted with questions that aim to identify the shared beliefs (represented in terms of the contents of the three banks) at each turn in the dialogue along with the current context. By treating CGT as a QA task, we provide a structured method for quantitatively evaluating the effectiveness of systems in tracking and updating shared beliefs among dialogue participants. This formulation not only helps in understanding the common ground reached but also in assessing the implicit and explicit acknowledgment of information as the conversation progresses.</p>
</sec>
<sec id="s5">
<title>5 Dataset</title>
<p>For our experiments, we use the Weights Task Dataset (WTD) (Khebour et al., <xref ref-type="bibr" rid="B41">2023</xref>, <xref ref-type="bibr" rid="B42">2024</xref>). The WTD contains ten videos, totaling &#x0007E;170 min, in which groups of three were asked to determine the weights of five blocks using a balance scale. During the task, participants communicated with each other using multiple modalities, including language, gesture, gaze, and action. Participants were recruited from a university setting, spoke English, and were between 19 and 35 years of age.</p>
<p>The WTD includes multiple layers of annotations. Speech was segmented and transcribed three ways: automatically, using Google Cloud ASR and Whisper; and manually by humans. Gestures, including deictic (pointing), iconic (depicting properties of objects or actions), and emblematic or conventional gestures, were annotated using Gesture AMR (GAMR) (Brutti et al., <xref ref-type="bibr" rid="B10">2022</xref>; Donatelli et al., <xref ref-type="bibr" rid="B27">2022</xref>). Actions, including participant actions (lifting blocks, or putting them on other objects) and scale actions (whether the scale is balanced, or leaning in some direction), were represented using VoxML (Pustejovsky and Krishnaswamy, <xref ref-type="bibr" rid="B65">2016</xref>). Collaborative problem-solving indicators, measuring ways in which groups share knowledge and skills to jointly solve problems, were annotated using the framework of Sun et al. (<xref ref-type="bibr" rid="B71">2020</xref>). The NICE coding scheme (Dey et al., <xref ref-type="bibr" rid="B24">2023</xref>) was used to annotate additional indicators of engagement, including gaze, posture, and emotion. Finally, the WTD contains Common Ground Annotations (CGA); these include dialogue moves, such as STATEMENT (announcement of some proposition), ACCEPT (agreement with a previous statement), and DOUBT (disagreement with a previous statement); and participant observations and inferences that justify statements.</p>
<sec>
<title>5.1 Common ground tracking in the weights task dataset</title>
<p>At the beginning of each Weights Task dialogue, we initialize QB<sc>ank</sc> with propositions, where each proposition states that a certain block (denoted by its color, red, blue, green, purple, or yellow) has a certain weight (between 10 and 50 grams, in 10-gram intervals). With five blocks and five possible weights, QB<sc>ank</sc> contains 5 &#x000D7; 5 &#x0003D; 25 propositions. Meanwhile EB<sc>ank</sc> and FB<sc>ank</sc> are initially empty, as nothing has yet been discussed.</p>
<p>As the dialogue progresses, we update the CGS as follows, according to the CGA. The STATEMENT of a proposition (e.g., <italic>blue is 10</italic>), or of something that would entail it (e.g., <italic>red and blue are equal</italic>, when <monospace>red</monospace> = <monospace>10</monospace> is already in FB<sc>ank</sc>), moves that proposition (<monospace>blue</monospace> = <monospace>10</monospace>) from QB<sc>ank</sc> to EB<sc>ank</sc>. An ACCEPT of that proposition (e.g., <italic>I agree</italic>) then moves it from EB<sc>ank</sc> to FB<sc>ank</sc>, and removes inconsistent propositions (e.g., <monospace>blue</monospace> = <monospace>20</monospace>, <monospace>blue</monospace> = <monospace>30</monospace>, etc.) from the CGS.</p>
<p>As an example, in <xref ref-type="fig" rid="F2">Figure 2</xref>, the participants have a shared belief that the blue block weighs 10 grams, while it is not yet common knowledge that the red block weighs 10 grams. In other words, <monospace>blue = 10</monospace> is in FB<sc>ank</sc>, while <monospace>red = 10</monospace> is in QB<sc>ank</sc>. After putting the blue and red blocks on the scale and observing that the scale is balanced, participant 1 says &#x0201C;Yeah OK so now we know that this is also ten&#x0201D;. This moves <monospace>red = 10</monospace> from QB<sc>ank</sc> to EB<sc>ank</sc>. Participant 2 then says &#x0201C;OK&#x0201D;; this promotes <monospace>red = 10</monospace> from EB<sc>ank</sc> to FB<sc>ank</sc>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Example of a common ground update in the Weights Task. <bold>(Left)</bold> P1 believes &#x0201C;blue=10g&#x0201D;, but does not agree that &#x0201C;red=10g.&#x0201D; <bold>(Right)</bold> After seeing the scale, P1, P2, and P3 all agree on both propositions.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1479905-g0002.tif"/>
</fig>
</sec>
</sec>
<sec id="s6">
<title>6 Experiments</title>
<p>In this section, we present experiments on the CGT task by applying our proposed MMDP pipeline (Section 6.2.4) on the Weights Task Dataset under a zero-shot learning setting. At a high level, we formalize CGT as a closed-domain QA task, where the language model is prompted with the evidential context from a dialogue segment and a question asking about the established common ground regarding the block weights. Based on the DP outputs, the context for each question also includes the natural language utterance paraphrases of all previous turns from the beginning of the dialogue. At each turn, the question includes the model prediction of the CG from the last dialogue segment (underscored text in <xref ref-type="fig" rid="F3">Figure 3</xref>).<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> We also instruct the model to generate the prediction in JSON format, so that it can be easily incorporated into the question prompt or processed for the evaluation. We experiment with GPT-3.5 (Brown et al., <xref ref-type="bibr" rid="B9">2020</xref>) for both the DP and QA steps for its accessibility and cost-efficiency. We use the OpenAI API version <monospace>gpt-3.5-turbo-0125</monospace>. Finally, we use the Dice Similarity Coefficient (DSC) as the evaluation metric (S&#x000F8;rensen, <xref ref-type="bibr" rid="B70">1948</xref>; Dice, <xref ref-type="bibr" rid="B25">1945</xref>). DSC is similar to F1 score, measuring the similarity between gold and predicted common ground propositions.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Common ground tracking pipeline with LLMs. Text format and emphasis on model input are addded for clarity.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1479905-g0003.tif"/>
</fig>
<sec>
<title>6.1 Design</title>
<p>We propose a new method, MMDP, that can improve the CGT task by utilizing language only LLMs. Instead of using Multimodal LLMs that consist of different encoders to encode information from multiple modalities (Yin et al., <xref ref-type="bibr" rid="B86">2023</xref>), we extend DP to the action and gesture annotations from the WTD. We leverage the capability of LLMs to paraphrase multimodal input into a natural language form, and infer the common ground from the dialogue context. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates our proposed LLM-prompting pipeline for modeling the CGT task. In the rest of this section, we first describe the data preprocessing pipeline (Section 6.2) where the DP techniques are used, and then the description of the prompt design and major components that use the paraphrases.</p>
</sec>
<sec>
<title>6.2 Data preprocessing pipeline</title>
<p>We describe the data selection and processing pipeline on the WTD to prepare conversational inputs to the model. The source of the annotations described in this section is a combination of Khebour et al. (<xref ref-type="bibr" rid="B41">2023</xref>) and Khebour et al. (<xref ref-type="bibr" rid="B42">2024</xref>). Using our preprocessing pipeline, we experiment with primarily two subtasks (anaphora resolution and multimodal alignment) of DP as the implementation of the proposed MMDP.</p>
<sec>
<title>6.2.1 Speech</title>
<p>The speech audio from the WTD is segmented into utterances delimited by silence. Each utterance is manually transcribed, and we refer to this set of text as &#x0201C;raw&#x0201D; utterances. In addition, to enhance the CGT performance of the LLM, we decontexualize pronouns of task-relevant entities in the dialogue through coreferential redescription. We believe this DP method of redescription can link the same entities across different modalities and serve as an alignment in our uni-modal system. Following our previous work (Rim et al., <xref ref-type="bibr" rid="B66">2023</xref>; Tu et al., <xref ref-type="bibr" rid="B78">2023</xref>), we paraphrase the mentions that refer to the same entity into their most informative form, i.e., proper nouns. In example 1, we paraphrase &#x0201C;that one&#x0201D; into &#x0201C;the blue block&#x0201D; for systems to better understand the context.</p>
<list list-type="simple">
<list-item><p>(1) <bold>P1 utterance</bold>: Maybe we would put that one there too.</p></list-item>
<list-item><p><bold>P1 utterance with DP</bold>: Maybe we would put <italic>the blue block</italic> there too.</p></list-item>
</list>
<p>This enriched set of text is referred to as &#x0201C;decontextualized&#x0201D; utterances in the rest of the paper. In our experiment, we use both the raw and decontextualized utterances, to measure the impact of DP.</p>
</sec>
<sec>
<title>6.2.2 Actions</title>
<p>The WTD provides manual annotations of agentive actions regarding block placement. The annotation is done in semi-logical, parenthesized form, but we found some annotation errors while experimenting. Hence we decided to review the entire action annotation, and manually fixed the found errors. Most of the errors we found were missing annotations when multiple blocks were moved together, but also a smaller number of duplicates and incorrect block color markings were found.</p>
</sec>
<sec>
<title>6.2.3 Gesture</title>
<p>We convert the gesture annotation from GAMR syntax to &#x0201C;enclosed&#x0201D; text with parentheses to mark up patterns that can be more efficiently interpreted by language models (Zhai et al., <xref ref-type="bibr" rid="B87">2022</xref>; Zhang et al., <xref ref-type="bibr" rid="B88">2023</xref>). This also made the syntax more consistent with the VoxML-based action annotations when aligned together. We adopt a heuristic method to map the gesture acts from the datasets to their closest event head (e.g., <monospace>deixis-GA</monospace> to <italic>point</italic>, <monospace>emblem-GA</monospace> to <italic>confirm</italic>), and parse the gesture graph to extract the corresponding arguments. Specifically, for example in 2, we map the deictic act to the pointing action, and remove the argument name and variable to keep it simple in the input.</p>
<list list-type="simple">
<list-item><p>(2) <bold>GAMR</bold>:</p></list-item>
<list-item><p>&#x000A0;&#x000A0;&#x000A0;&#x000A0;<monospace>(d / deixis-GA</monospace></p></list-item>
<list-item><p>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;<monospace>:ARG0 (p1 / participant_1)</monospace></p></list-item>
<list-item><p>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;<monospace>:ARG1 (b / blue_block)</monospace></p></list-item>
<list-item><p>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;<monospace>:ARG2 (g / group))</monospace></p></list-item>
<list-item><p>&#x000A0;&#x000A0;&#x000A0;&#x000A0;<bold>Enclosed:</bold></p></list-item>
<list-item><p>&#x000A0;&#x000A0;&#x000A0;&#x000A0;<monospace>point(blue_block,(other participants))</monospace></p></list-item>
</list>
</sec>
<sec>
<title>6.2.4 Multimodal alignment</title>
<p>Following the same setting in Khebour et al. (<xref ref-type="bibr" rid="B42">2024</xref>), we align the actions and gestures with the utterance that overlaps the most in terms of the starting and ending times. As briefly discussed in Section 3.2, we use two different forms of linguistic paraphrasing, the Machine Readable Paraphrase (MRP) and Human Readable Paraphrase (HRP), to obtain alignment of information across different modalities.</p>
<p>Specifically for this work, MRP is a form of (semi-)structured textual representation of the multimodality being expressed in the dialogue. Concretely, we generate an MRP of a multimodal dialogue segment as a set of key-value pairs that map each agent and modality to the content of the communicative event (e.g., action, utterance, gesture, etc.). While doing so, we apply some normalization to the raw annotation (Section 4). MRP features a uniform structure and text patterns that efficiently encode the semantics of the multimodal interactions in a dialogue. It also provides a pluggable expansibility for additional modalities, by adding or removing keyed pairs from the structure.</p>
<p>The second step of MMDP is the conversion from MRP to HRP with the application of LLMs. Compared to the MRP, the HRP in its natural language form is more effective to be processed and interpreted by language models. Similar to the paraphrases from DP, the HRP also encodes implicit semantics, enabled by LLMs&#x00027; capabilities to reconstruct sentence structures of the (often incomplete and disfluent) speech and to resolve anaphoric references across different modalities. This can help generate more coherent paraphrases. We show how HRP conversion is done and then show the utility of MMDP by applying it on WTD in the following sections.</p>
</sec>
<sec>
<title>6.2.5 Dialogue segmentation</title>
<p>In the CGT task, we focus on identifying the common ground that is updated right after the ACCEPT dialogue move. The ACCEPT move is essential in establishing the common ground in the whole dialogue, and previous work (Khebour et al., <xref ref-type="bibr" rid="B42">2024</xref>) finds that it is more challenging to model the ACCEPT move than the other moves. We split the dialogues into segments on the ending time of each ACCEPT move. We show the number of ACCEPT moves (segments) and utterances in <xref ref-type="table" rid="T1">Table 1</xref>. On average, each group is annotated with 4.5 ACCEPTs. The group with the most ACCEPTs has six segments and the least, 2. The average number of utterances in each group is 43.4 where group 7 has the most utterances (54) and group 9 has the least (19).</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Statistics of accepted statements and utterances in CGA.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="left"><bold>Count</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">&#x00023; of groups</td>
<td valign="top" align="center">10</td>
</tr> <tr>
<td valign="top" align="left">Avg. &#x00023; of utterance per group</td>
<td valign="top" align="center">43.4</td>
</tr> <tr>
<td valign="top" align="left">Min / max &#x00023; of utterances</td>
<td valign="top" align="center">19/54</td>
</tr> <tr>
<td valign="top" align="left">Avg. &#x00023; of ACCEPT moves per group</td>
<td valign="top" align="center">4.5</td>
</tr> <tr>
<td valign="top" align="left">Min / max &#x00023; of ACCEPT moves</td>
<td valign="top" align="center">2/6</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec>
<title>6.3 Experiments with Large Language Models</title>
<sec>
<title>6.3.1 In-context task instructions</title>
<p>We apply the LLMs on the CGT task under an in-context learning scenario. We first manually generate the Weights Task description of the situated task setting (red unit in <xref ref-type="fig" rid="F3">Figure 3</xref>), and use it as the system prompt input to the model. Within each segment of dialogue that establishes common ground, we create a prompt for each turn with the multimodal MRP that is converted from the existing annotations, and ask the model to generate an HRP in a natural language form. At the end of each dialogue segment, we instruct the model to infer the current common ground over the block weights by prompting it with the question.</p>
</sec>
<sec>
<title>6.3.2 Dense paraphrasing of multimodal input</title>
<p>As shown in <xref ref-type="fig" rid="F3">Figure 3</xref> (blue unit), given the aligned annotations, we create an MRP as a key-value pair structure, where the key encodes the speaker ID and the modality, and the value encodes the annotation contents, normalized for non-speech modalities (Section 4). This set of pairs is then serialized into a concatenated string representation, which we call MRP.</p>
<list list-type="simple">
<list-item><p>(3) <bold>P1 utterance</bold>: Maybe we would put that one there too.</p></list-item>
<list-item><p><bold>P1 gesture</bold>: <monospace>point(blue_block,(other participants))</monospace></p></list-item>
</list>
<p>Example 3 shows a sample utterance with an aligned gesture, transformed to an MRP. After the MRP is constructed, we apply the language model to convert it to an HRP (Section 6.2.4). In order to generate the HRP from each turn, the current MRP along with all the HRPs from previous turns starting from the beginning of the dialogue are included in the context prompt. <xref ref-type="fig" rid="F4">Figure 4</xref> shows the full prompt for the CGT pipeline. The data input is changed accrodingly to accommodate different experiment settings.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Full prompt to the LLMs for the common ground tracking pipeline. Text format and emphasis are added for clarity.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1479905-g0004.tif"/>
</fig>
</sec>
<sec>
<title>6.3.3 Baseline settings</title>
<p>We evaluate our approach against CGT baselines across three input settings: language-only, all-modalities in textual form, and all-modalities incorporating both text and images. For language-only and all-modalities in textual form, we employ baseline models from Khebour et al. (<xref ref-type="bibr" rid="B42">2024</xref>). In the language-only scenario, Khebour et al. (<xref ref-type="bibr" rid="B42">2024</xref>) transform decontextualized utterances (<sc>Decont.</sc>) into embeddings and utilize a similarity-based method to identify the common ground. For the all-modalities in textual form setting, a hybrid method is used which involves human annotations to map predicted utterance IDs to the corresponding common ground.</p>
<p>In addition to textual input, our method capitalizes on LLMs to reason with both text and images. Specifically, we extract five image frames evenly from each utterance&#x00027;s corresponding video clip and use these frames together with the utterances as input to incorporate multimodal information.<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref> For this setting, we apply GPT-4o and GPT-4o-mini as baseline models.</p>
</sec>
</sec>
<sec>
<title>6.4 Results</title>
<p><xref ref-type="table" rid="T2">Table 2</xref> compares the CGT results between the baseline models and our methods under different settings. Under the language-only setting, <sc>DP-Utt.</sc> and <sc>DP-Decont.</sc> use raw and decontextualized utterances, respectively, in our pipeline without the paraphrasing step. Compared to the baseline results that use the decontextualized utterances as input, <sc>DP-Utt.</sc> is able to achieve comparable results (0.6 points lower) without access to the decontextualized information, suggesting LLMs are better at learning from the conversation context. However, by using the same decontextualized utterances as the input, <sc>DP-Decont.</sc> outperforms the baseline by a large margin (20.4 points).</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Evaluation results on the CGT task.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Group 1</bold></th>
<th valign="top" align="center"><bold>Group 2</bold></th>
<th valign="top" align="center"><bold>Group 3</bold></th>
<th valign="top" align="center"><bold>Group 4</bold></th>
<th valign="top" align="center"><bold>Group 5</bold></th>
<th valign="top" align="center"><bold>Group 6</bold></th>
<th valign="top" align="center"><bold>Group 7</bold></th>
<th valign="top" align="center"><bold>Group 8</bold></th>
<th valign="top" align="center"><bold>Group 9</bold></th>
<th valign="top" align="center"><bold>Group 10</bold></th>
<th valign="top" align="center"><bold>Avg</bold>.</th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="12"><bold>Language-only</bold></td>
</tr> <tr>
<td valign="top" align="left">B<sc>aseline</sc> (Khebour et al., <xref ref-type="bibr" rid="B42">2024</xref>)</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">52.8</td>
<td valign="top" align="center"><bold>50.1</bold></td>
<td valign="top" align="center">4.5</td>
<td valign="top" align="center">16.5</td>
<td valign="top" align="center"><bold>37.2</bold></td>
<td valign="top" align="center"><bold>82.5</bold></td>
<td valign="top" align="center">52.6</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">29.6</td>
</tr> <tr>
<td valign="top" align="left">DP-U<sc>tt.</sc></td>
<td valign="top" align="center">74.8</td>
<td valign="top" align="center">39.9</td>
<td valign="top" align="center">37.5</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">56.1</td>
<td valign="top" align="center">6.1</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">47.4</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">28.6</td>
<td valign="top" align="center">29.0</td>
</tr> <tr>
<td valign="top" align="left">DP-D<sc>econt.</sc></td>
<td valign="top" align="center"><bold>90.1</bold></td>
<td valign="top" align="center"><bold>58.0</bold></td>
<td valign="top" align="center">35.8</td>
<td valign="top" align="center"><bold>45.0</bold></td>
<td valign="top" align="center"><bold>66.7</bold></td>
<td valign="top" align="center">17.2</td>
<td valign="top" align="center">61.8</td>
<td valign="top" align="center"><bold>60.6</bold></td>
<td valign="top" align="center"><bold>3.6</bold></td>
<td valign="top" align="center"><bold>60.9</bold></td>
<td valign="top" align="center"><bold>50.0</bold></td>
</tr> <tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="12"><bold>All-modalities in textual form</bold></td>
</tr> <tr>
<td valign="top" align="left">B<sc>aseline</sc> (Khebour et al., <xref ref-type="bibr" rid="B42">2024</xref>)</td>
<td valign="top" align="center">42.5</td>
<td valign="top" align="center">48.0</td>
<td valign="top" align="center"><bold>41.8</bold></td>
<td valign="top" align="center">34.8</td>
<td valign="top" align="center">31.8</td>
<td valign="top" align="center"><bold>31.5</bold></td>
<td valign="top" align="center"><bold>63.7</bold></td>
<td valign="top" align="center">57.4</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center"><bold>79.4</bold></td>
<td valign="top" align="center">43.1</td>
</tr> <tr>
<td valign="top" align="left">MMDP-U<sc>tt.</sc></td>
<td valign="top" align="center">85.0</td>
<td valign="top" align="center">36.5</td>
<td valign="top" align="center">37.5</td>
<td valign="top" align="center">38.2</td>
<td valign="top" align="center">54.3</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">55.2</td>
<td valign="top" align="center">48.7</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">63.7</td>
<td valign="top" align="center">41.9</td>
</tr> <tr>
<td valign="top" align="left">MMDP-D<sc>econt.</sc></td>
<td valign="top" align="center"><bold>88.3</bold></td>
<td valign="top" align="center"><bold>58.0</bold></td>
<td valign="top" align="center">35.8</td>
<td valign="top" align="center"><bold>45.0</bold></td>
<td valign="top" align="center"><bold>65.2</bold></td>
<td valign="top" align="center">17.2</td>
<td valign="top" align="center">55.2</td>
<td valign="top" align="center"><bold>63.3</bold></td>
<td valign="top" align="center"><bold>13.8</bold></td>
<td valign="top" align="center">71.6</td>
<td valign="top" align="center"><bold>51.3</bold></td>
</tr> <tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="12"><bold>All-modalities in text and video frames</bold></td>
</tr> <tr>
<td valign="top" align="left"><sc>Baseline-GPT-4o-mini</sc></td>
<td valign="top" align="center">55.3</td>
<td valign="top" align="center">33.1</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">32.1</td>
<td valign="top" align="center">38.0</td>
<td valign="top" align="center">26.5</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">48.7</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">31.0</td>
<td valign="top" align="center">26.5</td>
</tr> <tr>
<td valign="top" align="left"><sc>Baseline-GPT-4o</sc></td>
<td valign="top" align="center">84.1</td>
<td valign="top" align="center">33.1</td>
<td valign="top" align="center">34.0</td>
<td valign="top" align="center">32.1</td>
<td valign="top" align="center">47.8</td>
<td valign="top" align="center">26.5</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">43.3</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">43.8</td>
<td valign="top" align="center">34.5</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>DSC is reported for each group and the average under multimodal and language only settings. The bold value indicates the best DSC under different settings.</p>
</table-wrap-foot>
</table-wrap>
<p>Under the setting of all-modalities in textual form, the B<sc>aseline</sc> adopts a hybrid method that uses annotations to map predicted utterance IDs to the corresponding common ground. <sc>MMDP-Utt.</sc> combines the action, gesture and raw utterance in the MRP as the input. Similarly, <sc>MMDP-Decont.</sc> uses the decontextualized utterance in the MRP instead. Compared to <sc>DP-Utt.</sc>, <sc>MMDP-Utt.</sc> improves the results by 13 points, suggesting the usefulness of multimodal information for the CGT task. Both <sc>DP-Decont.</sc> (6.9 points) and <sc>MMDP-Decont.</sc> (8.2 points) perform better than the stronger multimodal baseline. Compared to <sc>DP-Decont.</sc>, <sc>MMDP-Decont.</sc> performs only slightly better by incorporating additional annotations from other modalities (1.3 points). This may suggest that the decontextualized utterances have already encoded most of the multimodal information, and <sc>MMDP-Decont.</sc> exhibits an upper-bound performance for the CGT task.</p>
<p>Compared to representing multimodalties in MRP, using video frames as additional input does not exhibit better performance over our proposed method. Under this setting, baseline with GPT-4o outperforms GPT-4o-mini, yet it is still worse than <sc>MMDP-Utt.</sc> which integrates action, gesture and raw utterance in the MRP (7.4 points lower). Overall, we show the effectiveness of our LLM pipeline, and the decontextualized utterances enhanced with multimodal textual paraphrases can yield the best results for the task.</p>
<sec>
<title>6.4.1 Error analysis</title>
<p>While the dialogues are all about the Weights Task in the dataset, the conversations from different groups exhibit various patterns that are also reflected in the CGT results. We briefly characterize the cases where the performance from the baselines and our methods have salient gaps on individual groups.</p>
<p>The MMDP method improves the most on group 1 (90.1 points for language only, 45.8 points for all modalities). By examining the dialogue, we find that this group builds up the common ground in a &#x0201C;bottom-up&#x0201D; style by identifying the block weights from the lightest to the heaviest. This way the conversation depends heavily on the context, making MMDP a better choice to capture these long dependencies. In addition, all modalities in this group play important roles in identifying the common ground.</p>
<list list-type="simple">
<list-item><p>(4) <bold>P2 utterance</bold>: That&#x00027;s <inline-formula><mml:math id="M1"><mml:mrow><mml:mstyle mathcolor="blue"><mml:mtext>ten</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula> so then</p></list-item>
<list-item><p><bold>P1 action</bold>: <monospace>put(<inline-formula><mml:math id="M2"><mml:mrow><mml:mstyle mathvariant="monospace" mathcolor="blue"><mml:mtext>blue_block</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula>,(left_scale))</monospace></p></list-item>
<list-item><p><bold>Common ground</bold>: <monospace>blue = 10</monospace></p></list-item>
</list>
<list list-type="simple">
<list-item><p>(5) <bold>P2 utterance</bold>: Probably <inline-formula><mml:math id="M3a"><mml:mrow><mml:mstyle mathcolor="#bf0040;"><mml:mtext>thirty</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula> at this point</p></list-item>
<list-item><p><bold>P1 action</bold>: <monospace>point</monospace>(<inline-formula><mml:math id="M4"><mml:mrow><mml:mstyle mathvariant="monospace" mathcolor="#bf0040;"><mml:mtext>purple_block</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula><monospace>,(other participants))</monospace></p></list-item>
<list-item><p><bold>Common ground</bold>: <monospace>purple = 30</monospace></p></list-item>
</list>
<p>Consider example 4. The utterance from Participant 2 mentions the possible weight of a block, and the aligned putting action from Participant 1 indicates that the block is blue. Similarly in example 5, The pointing gesture also indicates the weight from the utterance is for the purple block.</p>
<p>Although our method improves the overall performance, the baseline performs better on group 6 (20 points for language only, 14.3 points for all modalities). Unlike group 1, we observe that the dialogue from this group contains many implicit assumptions that are not expressed either verbally or non-verbally. This makes the annotation quite sparse and difficult for LLMs to build up the conclusion from the context. This pattern also appears in group 3. Participants also sometimes refer to the color of the block in a non-standardized way, which causes further confusion for the model.</p>
<list list-type="simple">
<list-item><p>(6) <bold>P2 utterance</bold>: So <inline-formula><mml:math id="M51a"><mml:mrow><mml:mstyle mathcolor="#bf0040;"><mml:mtext>big&#x000A0;blue</mml:mtext></mml:mstyle></mml:mrow></mml:math></inline-formula> is probably thirty</p></list-item>
<list-item><p><bold>Common ground</bold>: <monospace>purple = 30</monospace></p></list-item>
</list>
<p>In example 6, participant 2 refers to the color of the purple block as &#x0201C;big blue&#x0201D; throughout the whole dialogue.</p>
<p>CGT on the dialogue from group 9 is challenging to both the baseline and MMDP. After examining the data, we notice that most action and gesture annotations are not aligned with the utterances, making the improvement from multimodal information incremental. This may be due to the nature of the conversation where non-verbal actions happen asynchronously with the utterance. In addition, the less frequent usage of pronoun references in this dialogue makes it difficult to take advantage of the decontexualization of the utterances.</p>
<list list-type="simple">
<list-item><p>(7) <bold>P3 utterance</bold>: Looks equal yeah</p></list-item>
<list-item><p><bold>P2 utterance</bold>: Yeah that&#x00027;s good</p></list-item>
<list-item><p><bold>P1 utterance</bold>: Look we have the thirty gram block</p></list-item>
</list>
<p>Example 7 shows the key utterances for establishing the common ground from group 9. The lack of proper multimodal alignments and block references poses a lot of challenges to the CGT automation.</p>
<p>Multuimodal GPT with both text and image input performs worse than textual MRP and HRP. This could be attributed to the insufficient salient mappings between videos frames and the corresponding utterance. Notably in Group 7, where the models struggle to identify the correct common grounds, many actions (e.g., <italic>slightly lift the block and then put it back on the scale</italic>) involve quick and subtle movements that are challenging for the models to accurately capture. Moreover, gestures in the video can be inherently ambiguous, especially when a participant points to a specific block that is positioned near other blocks. However, the converted MRP from the multimodal input is useful in providing accurate information and eliminating the ambiguities from the video frames.</p>
</sec>
</sec>
</sec>
<sec id="s7">
<title>7 Discussion and analysis of MMDP</title>
<p>In this section, we further explore the utility of the MMDP method. We experiment with MMDP on the CGT task, and conduct quantitative analysis of the results with different model selection and input data variance.</p>
<sec>
<title>7.1 Larger language models</title>
<p>We evaluate a larger and more powerful language model in the MMDP pipeline. We apply GPT-4o (OpenAI, <xref ref-type="bibr" rid="B59">2023</xref>) for both the DP and QA steps. We use the OpenAI API with version <monospace>gpt-4o-2024-05-13</monospace>. <xref ref-type="table" rid="T3">Table 3</xref> shows the model comparison results. Overall, GPT-4o performs better than GPT-3.5 when decontextualized or multimodal information is provided in the input. However, GPT-4o does not show superior results on the <monospace>DP-Utt.</monospace> setting. This confirms our findings that the richness of the multimodal information is essential to resolve the CGT task.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Evaluation results on the CGT task.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>DP-Utt</bold>.</th>
<th valign="top" align="center"><bold>DP-Decont</bold>.</th>
<th valign="top" align="center"><bold>MMDP-Utt</bold>.</th>
<th valign="top" align="center"><bold>MMDP-Decont</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">GPT-3.5</td>
<td valign="top" align="center"><bold>29.0</bold></td>
<td valign="top" align="center">50.0</td>
<td valign="top" align="center">41.9</td>
<td valign="top" align="center">51.3</td>
</tr> <tr>
<td valign="top" align="left">GPT-4o</td>
<td valign="top" align="center">28.6</td>
<td valign="top" align="center"><bold>53.8</bold></td>
<td valign="top" align="center"><bold>45.8</bold></td>
<td valign="top" align="center"><bold>54.9</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>We compare GPT-3.5 with GPT-4o under different pipeline settings. Average DSC over all groups is reported. The bold value indicates the best DSC under different settings.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>7.2 Multimodal information encoded with HRP</title>
<p>In the MMDP pipeline, we propose a DP step that converts the multimodal MRP into HRP. We explore the utility of the DP step by using MRP vs. HRP as the model input. <xref ref-type="table" rid="T4">Table 4</xref> shows the evaluation results. In general, models with HRP perform better than those with MRP, suggesting the effectiveness of DP in grounding non-verbal information into language form. Compared to GPT-3.5, applying DP with GPT-4o results in less differentiation in the performance (3.1 vs. 7.5). This indicates that a larger language model has more capabilities to learn structured information from MRP directly.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Evaluation results from the GPT models under the multimodal setting.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Setting</bold></th>
<th valign="top" align="center"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Use HRP</bold></th>
<th valign="top" align="center"><bold>DSC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="4">MMDP-Utt.</td>
<td valign="top" align="center">GPT-3.5</td>
<td valign="top" align="center">&#x02718;</td>
<td valign="top" align="center">34.4</td>
</tr>
 <tr>
<td valign="top" align="center">GPT-3.5</td>
<td valign="top" align="center">&#x02714;</td>
<td valign="top" align="center">41.9</td>
</tr>
 <tr>
<td valign="top" align="center">GPT-4o</td>
<td valign="top" align="center">&#x02718;</td>
<td valign="top" align="center">42.7</td>
</tr>
 <tr>
<td valign="top" align="center">GPT-4o</td>
<td valign="top" align="center">&#x02714;</td>
<td valign="top" align="center">45.8</td>
</tr> <tr>
<td valign="top" align="left">Baseline<sup>&#x02021;</sup></td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">43.1</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">MMDP-Decont.</td>
<td valign="top" align="center">GPT-3.5</td>
<td valign="top" align="center">&#x02718;</td>
<td valign="top" align="center">47.3</td>
</tr>
 <tr>
<td valign="top" align="center">GPT-3.5</td>
<td valign="top" align="center">&#x02714;</td>
<td valign="top" align="center">51.3</td>
</tr>
 <tr>
<td valign="top" align="center">GPT-4o</td>
<td valign="top" align="center">&#x02718;</td>
<td valign="top" align="center">52.9</td>
</tr>
 <tr>
<td valign="top" align="center">GPT-4o</td>
<td valign="top" align="center">&#x02714;</td>
<td valign="top" align="center">54.9</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>We compare the DSC with or without the DP step for HRP generation. <sup>&#x02021;</sup>Baseline from all modalities.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>7.3 Dialogue context cutoff</title>
<p>We evaluate whether MMDP can enable more efficient learning by cutting off the previous dialogue context in the input. In our current pipeline, in the prompt for every DP and QA step, we include previous generated HRPs and common ground predictions from the <italic>beginning</italic> of the dialogue. In this experiment, we only keep the HRPs from the <italic>current</italic> dialogue segment in the prompt. <xref ref-type="table" rid="T5">Table 5</xref> shows the evaluation results. In general, we notice a performance drop under most settings after applying the context cutoff. Although the question prompt still has access to the previous common ground prediction, the limited context poses additional challenges to the model. <sc>MMDP-Decont.</sc> has the highest drop (6.2) in performance. This may be because the combination of decontextualized utterance and multimodal information from the bigger context contributes the most to model performance. <sc>DP-Utt.</sc> shows a similar result with the cutoff. This may result from the already existing lack of annotation in the context of raw utterances. Overall, we observe that although there exists a trade-off between performance and efficiency, the model with context cutoff is still able to produce competitive results compared to the baseline (43.1).</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Evaluation results on the CGT task.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>DP-Utt</bold>.</th>
<th valign="top" align="center"><bold>DP-Decont</bold>.</th>
<th valign="top" align="center"><bold>MMDP-Utt</bold>.</th>
<th valign="top" align="center"><bold>MMDP-Decont</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Cutoff</td>
<td valign="top" align="center">28.7</td>
<td valign="top" align="center">50.3</td>
<td valign="top" align="center">43.9</td>
<td valign="top" align="center">48.7</td>
</tr> <tr>
<td valign="top" align="left">No-cutoff</td>
<td valign="top" align="center">28.6</td>
<td valign="top" align="center">53.8</td>
<td valign="top" align="center">45.8</td>
<td valign="top" align="center">54.9</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>We apply GPT-4o and compare the average DSC with or without context cutoff.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>7.4 Re-annotation of CGA</title>
<p>Since the size of the CGA is limited, we provide additional annotations for future research. Specifically, in our experiments, we find that STATEMENTs are often not followed by explicit ACCEPTs. This results in propositions remaining in EB<sc>ank</sc> and not moving to FB<sc>ank</sc>, even when the dialogue continues as if the participants all believe the stated proposition. For this reason, we add an implicit ACCEPT to each STATEMENT in the CGA, except those that are followed by a DOUBT. This can be seen as allowing most STATEMENTs to directly promote propositions from QB<sc>ank</sc> to FB<sc>ank</sc>. The re-annotation increases the average number of ACCEPTs from 4 to 14. The smallest increase is from 2 to 7 ACCEPTs. The most significant increase is observed in Group 5 that raises the number of ACCEPTs from 3 to 17. <xref ref-type="table" rid="T6">Table 6</xref> shows the number of ACCEPTs in the original and re-annotation of CGA.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Number of ACCEPTs in the original and re-annotation of CGA.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Original</bold></th>
<th valign="top" align="center"><bold>Re-annotation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Group 1</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">15</td>
</tr> <tr>
<td valign="top" align="left">Group 2</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">16</td>
</tr> <tr>
<td valign="top" align="left">Group 3</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">16</td>
</tr> <tr>
<td valign="top" align="left">Group 4</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">7</td>
</tr> <tr>
<td valign="top" align="left">Group 5</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">18</td>
</tr> <tr>
<td valign="top" align="left">Group 6</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">17</td>
</tr> <tr>
<td valign="top" align="left">Group 7</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">10</td>
</tr> <tr>
<td valign="top" align="left">Group 8</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">16</td>
</tr> <tr>
<td valign="top" align="left">Group 9</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">11</td>
</tr> <tr>
<td valign="top" align="left">Group 10</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">20</td>
</tr> <tr>
<td valign="top" align="left">All</td>
<td valign="top" align="center">45</td>
<td valign="top" align="center">146</td>
</tr></tbody>
</table>
</table-wrap>
<p>We run the same experiments on the new CGA data using GPT-3.5. <xref ref-type="table" rid="T7">Table 7</xref> shows the results. Although not directly comparable because of the different number of ACCEPTs, we notice that the average DSC on the re-annotated data is over 20 points higher than that on the original dataset. The results improve the most under the <sc>DP-Decont.</sc> setting (32.2 points higher). Overall, we find that using a less strict rule to identify ACCEPTs, and as a result, more accepted statements can lead to significant improvements on the CGT task. We suspect that the improvements stem from more ACCEPTs that agree with the same STATEMENT being annotated; e.g., there is only one ACCEPT of STATEMENT <monospace>red</monospace> = <monospace>10</monospace> in the original data. In the new data, two more ACCEPTs of the STATEMENT are annotated without any additional ACCEPTs to the other STATEMENTs.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Evaluation results from GPT-3.5 on the CGT task with re-annotated CGA.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>DP-Utt</bold>.</th>
<th valign="top" align="center"><bold>DP-Decont</bold>.</th>
<th valign="top" align="center"><bold>MMDP-Utt</bold>.</th>
<th valign="top" align="center"><bold>MMDP-Decont</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Original</td>
<td valign="top" align="center">29.0</td>
<td valign="top" align="center">50.0</td>
<td valign="top" align="center">41.9</td>
<td valign="top" align="center">51.3</td>
</tr> <tr>
<td valign="top" align="left">Re-annotation</td>
<td valign="top" align="center">56.4</td>
<td valign="top" align="center">82.1</td>
<td valign="top" align="center">67.3</td>
<td valign="top" align="center">75.5</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Average DSC over all the groups are reported.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>7.5 Limitations</title>
<p>One limitation of our work comes from the dataset selection, as our study of the CGT is solely based on the Weights Task Dataset (WTD). WTD contains ten recorded dialogues in a controlled setting, where three participants collaborate on a weight task to reach common ground. While the WTD provides a detailed view for examining human interactions over multiple communication modes, it may not fully capture the diversity found in real-world situations. Due to the small size of the dataset and the controlled task setting, the effectiveness of our MMDP method in understanding and tracking common ground may not easily extend to interactions that differ significantly from those in the WTD. To our best knowledge, WTD is the only exisiting CGT dataset. Future work could focus on expanding the dataset size and incorporating more diverse dialogues within other problem-solving task settings, such as tangram puzzles. Our experiments on the WTD involve dialogues in English only. Future studies involve exploring CGT in multilingual contexts.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s8">
<title>8 Conclusion</title>
<p>In this work, we have highlighted the importance of integrating multimodal representations in the development of more sophisticated and accurate dialogue systems, particularly in the service of addressing underspecified references within cross-modal settings. We proposed MMDP by extending the technique of DP for converting the annotations from multiple modalities into textual paraphrases with both machine-readable and human-readable formats. We built an LLM-based pipeline by applying MMDP on WTD, and showed that the generated paraphrases can be used effectively to improve performance on the CGT task under different model settings. We conducted a quantitative analysis of the results from experiments with different models, paraphrase input and context length, and showed that MMDP could still show competitive performance even with limited information from the input. We believe that MMDP for enhancing the interpretative power of multimodal dialogue systems constitutes a step toward a more capable and competent human-computer interaction in multimodal environments.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s9">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s10">
<title>Author contributions</title>
<p>JT: Conceptualization, Formal analysis, Investigation, Methodology, Resources, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. KR: Conceptualization, Data curation, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. BY: Investigation, Methodology, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. KL: Formal analysis, Resources, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. JP: Conceptualization, Funding acquisition, Investigation, Project administration, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s11">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This material is based in part upon work supported by Other Transaction award HR00112490377 from the U.S. Defense Advanced Research Projects Agency (DARPA) Friction for Accountability in Conversational Transactions (FACT) program; NSF National AI Institute for Student-AI Teaming (iSAT) under grant DRL 2019805; and NSF grant 2326985.</p>
</sec>
<ack>
<p>We would like to thank Nikhil Krishnaswamy, Ibrahim Khebour, Kelsey Sikes, Mariah Bradford, Brittany Cates, Paige Hansen, Changsoo Jung, Brett Wisniewski, Corbyn Terpstra, and Nathaniel Blanchard, at Colorado State University (CSU), Indrani Dey and Sadhana Puntambekar at University of Wisconsin Madison, Rachel Dickler and Leanne Hirshfield at University of Colorado, who were the co-creators of the Weights Task Dataset. We would also like to thank Nikhil Krishnaswamy, Ibrahim Khebour, Mariah Bradford, Benjamin Ibarra, and Nathaniel Blanchard for their work on the common ground tracking task. This material is based in part upon work supported by Other Transaction award HR00112490377 from the U.S. Defense Advanced Research Projects Agency (DARPA) Friction for Accountability in Conversational Transactions (FACT) program; NSF National AI Institute for Student-AI Teaming (iSAT) under grant DRL 2019805; and NSF grant 2326985. Approved for public release, distribution unlimited. Views expressed herein do not reflect the policy or position of the Department of Defense or the U.S. Government.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/brandeis-llc/mmdp-cgt.git">https://github.com/brandeis-llc/mmdp-cgt.git</ext-link></p></fn>
<fn id="fn0002"><p><sup>2</sup>An exception is the question for the first dialogue segment, for which there is no previous prediction.</p></fn>
<fn id="fn0003"><p><sup>3</sup>The average video clip length corresponding to each utterance is 4.3 s, with the longest being 21, 18, and 13 s, respectively. We believe that using five frames per utterance effectively captures the action and event dynamics occurring within the duration of each utterance.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Achiam</surname> <given-names>J.</given-names></name> <name><surname>Adler</surname> <given-names>S.</given-names></name> <name><surname>Agarwal</surname> <given-names>S.</given-names></name> <name><surname>Ahmad</surname> <given-names>L.</given-names></name> <name><surname>Akkaya</surname> <given-names>I.</given-names></name> <name><surname>Aleman</surname> <given-names>F. L.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>GPT-4 technical report</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2303.08774</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Angeli</surname> <given-names>G.</given-names></name> <name><surname>Manning</surname> <given-names>C. D.</given-names></name></person-group> (<year>2014</year>). <article-title>&#x0201C;NaturalLI: natural logic inference for common sense reasoning,&#x0201D;</article-title> in <source>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)</source>, eds. A. Moschitti, B. Pang, and W. Daelemans (Doha: Association for Computational Linguistics), <fpage>34</fpage>&#x02013;<lpage>545</lpage>. <pub-id pub-id-type="pmid">38753505</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asher</surname> <given-names>N.</given-names></name> <name><surname>Gillies</surname> <given-names>A.</given-names></name></person-group> (<year>2003</year>). <article-title>Common ground, corrections, and coordination</article-title>. <source>Argumentation</source> <volume>17</volume>, <fpage>481</fpage>&#x02013;<lpage>512</lpage>. <pub-id pub-id-type="doi">10.1023/A:1026346605477</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baltru&#x00161;aitis</surname> <given-names>T.</given-names></name> <name><surname>Ahuja</surname> <given-names>C.</given-names></name> <name><surname>Morency</surname> <given-names>L.-P.</given-names></name></person-group> (<year>2018</year>). <article-title>Multimodal machine learning: a survey and taxonomy</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>41</volume>, <fpage>423</fpage>&#x02013;<lpage>443</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2018.2798607</pub-id><pub-id pub-id-type="pmid">29994351</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barzilay</surname> <given-names>R.</given-names></name> <name><surname>Elhadad</surname> <given-names>M.</given-names></name></person-group> (<year>1997</year>). <article-title>&#x0201C;Using lexical chains for text summarization,&#x0201D;</article-title> in <source>Intelligent Scalable Text Summarization</source>, <fpage>111</fpage>&#x02013;<lpage>121</lpage>. <pub-id pub-id-type="pmid">18402049</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bateman</surname> <given-names>J.</given-names></name> <name><surname>Henschel</surname> <given-names>R.</given-names></name></person-group> (<year>1999</year>). <article-title>&#x0201C;From full generation to &#x02018;near-templates&#x00027; without losing generality,&#x0201D;</article-title> in <source>Proceedings of the KI&#x00027;99 Workshop &#x02018;May I Speak Freely&#x00027;</source> (<publisher-loc>Bonn</publisher-loc>), <fpage>13</fpage>&#x02013;<lpage>18</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhagat</surname> <given-names>R.</given-names></name> <name><surname>Hovy</surname> <given-names>E.</given-names></name></person-group> (<year>2013</year>). <article-title>What is a paraphrase?</article-title> <source>Comp. Linguist</source>. <volume>39</volume>, <fpage>463</fpage>&#x02013;<lpage>472</lpage>. <pub-id pub-id-type="doi">10.1162/COLI_a_00166</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boyer</surname> <given-names>M.</given-names></name> <name><surname>Lapalme</surname> <given-names>G.</given-names></name></person-group> (<year>1985</year>). <article-title>Generating paraphrases from meaning-text semantic networks</article-title>. <source>Comp. Intell</source>. <volume>1</volume>, <fpage>103</fpage>&#x02013;<lpage>117</lpage>. <pub-id pub-id-type="doi">10.1111/j.1467-8640.1985.tb00063.x</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>T.</given-names></name> <name><surname>Mann</surname> <given-names>B.</given-names></name> <name><surname>Ryder</surname> <given-names>N.</given-names></name> <name><surname>Subbiah</surname> <given-names>M.</given-names></name> <name><surname>Kaplan</surname> <given-names>J. D.</given-names></name> <name><surname>Dhariwal</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Language models are few-shot learners,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, Vol. 33</source>, eds. H. Larochelle, M. Ranzato, R. Hadsell, M. Balcan, and H. Lin (La Jolla, CA: Curran Associates, Inc.), <fpage>1877</fpage>&#x02013;<lpage>1901</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Brutti</surname> <given-names>R.</given-names></name> <name><surname>Donatelli</surname> <given-names>L.</given-names></name> <name><surname>Lai</surname> <given-names>K.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Abstract meaning representation for gesture,&#x0201D;</article-title> in <source>Proceedings of the Thirteenth Language Resources and Evaluation Conference</source> (<publisher-loc>Marseille</publisher-loc>: <publisher-name>European Language Resources Association</publisher-name>), <fpage>1576</fpage>&#x02013;<lpage>1583</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Budzianowski</surname> <given-names>P.</given-names></name> <name><surname>Wen</surname> <given-names>T.-H.</given-names></name> <name><surname>Tseng</surname> <given-names>B.-H.</given-names></name> <name><surname>Casanueva</surname> <given-names>I.</given-names></name> <name><surname>Ultes</surname> <given-names>S.</given-names></name> <name><surname>Ramadan</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;MultiWOZ - a large-scale multi-domain Wizard-of-Oz dataset for task-oriented dialogue modelling,&#x0201D;</article-title> in <source>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</source>, eds. E. Riloff, D. Chiang, J. Hockenmaier, and J. Tsujii (Brussels: Association for Computational Linguistics), <fpage>5016</fpage>&#x02013;<lpage>5026</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Busemann</surname> <given-names>S.</given-names></name> <name><surname>Horacek</surname> <given-names>H.</given-names></name></person-group> (<year>1998</year>). <article-title>A flexible shallow approach to text generation</article-title>. <source>arXiv</source> [preprint].</citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Byron</surname> <given-names>D. K.</given-names></name></person-group> (<year>2002</year>). <source>Resolving Pronominal Reference to Abstract Entities</source>. <publisher-loc>Rochester, NY</publisher-loc>: <publisher-name>University of Rochester</publisher-name>.</citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chai</surname> <given-names>H.</given-names></name> <name><surname>Moosavi</surname> <given-names>N. S.</given-names></name> <name><surname>Gurevych</surname> <given-names>I.</given-names></name> <name><surname>Strube</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Evaluating coreference resolvers on community-based question answering: from rule-based to state of the art,&#x0201D;</article-title> in <source>CRAC</source> (<publisher-loc>Stroudsburg, PA</publisher-loc>).</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Q.</given-names></name> <name><surname>Zhu</surname> <given-names>X.</given-names></name> <name><surname>Ling</surname> <given-names>Z.-H.</given-names></name> <name><surname>Wei</surname> <given-names>S.</given-names></name> <name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Inkpen</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Enhanced LSTM for natural language inference,&#x0201D;</article-title> in <source>Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</source>, eds. R. Barzilay, and M.-Y. Kan (Vancouver, BC: Association for Computational Linguistics), <fpage>1657</fpage>&#x02013;<lpage>1668</lpage>. <pub-id pub-id-type="pmid">37339033</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chhabra</surname> <given-names>A.</given-names></name> <name><surname>Vishwakarma</surname> <given-names>D. K.</given-names></name></person-group> (<year>2023</year>). <article-title>A literature survey on multimodal and multilingual automatic hate speech identification</article-title>. <source>Multim. Syst</source>. <volume>29</volume>, <fpage>1203</fpage>&#x02013;<lpage>1230</lpage>. <pub-id pub-id-type="doi">10.1007/s00530-023-01051-8</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Choi</surname> <given-names>E.</given-names></name> <name><surname>Palomaki</surname> <given-names>J.</given-names></name> <name><surname>Lamm</surname> <given-names>M.</given-names></name> <name><surname>Kwiatkowski</surname> <given-names>T.</given-names></name> <name><surname>Das</surname> <given-names>D.</given-names></name> <name><surname>Collins</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Decontextualization: making sentences stand-alone</article-title>. <source>Transact. Assoc. Comp. Linguist</source>. <volume>9</volume>, <fpage>447</fpage>&#x02013;<lpage>461</lpage>. <pub-id pub-id-type="doi">10.1162/tacl_a_00377</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chuang</surname> <given-names>Y.-S.</given-names></name> <name><surname>Liu</surname> <given-names>C.-L.</given-names></name> <name><surname>Lee</surname> <given-names>H.-Y.</given-names></name> <name><surname>shan Lee</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>Speechbert: an audio-and-text jointly learned language model for end-to-end spoken question answering</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.21437/Interspeech.2020-1570</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>H. H.</given-names></name> <name><surname>Brennan</surname> <given-names>S. E.</given-names></name></person-group> (<year>1991</year>). <source>Grounding in Communication</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Psychological Association</publisher-name>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Culicover</surname> <given-names>P. W.</given-names></name></person-group> (<year>1968</year>). <article-title>Paraphrase generation and information retrieval from stored text</article-title>. <source>Mech. Transl. Comput. Linguistics</source> <volume>11</volume>, <fpage>78</fpage>&#x02013;<lpage>88</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Das</surname> <given-names>R.</given-names></name> <name><surname>Singh</surname> <given-names>T. D.</given-names></name></person-group> (<year>2023</year>). <article-title>Multimodal sentiment analysis: a survey of methods, trends, and challenges</article-title>. <source>ACM Comp. Surv</source>. <volume>55</volume>, <fpage>1</fpage>&#x02013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.1145/3586075</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Del Tredici</surname> <given-names>M.</given-names></name> <name><surname>Shen</surname> <given-names>X.</given-names></name> <name><surname>Barlacchi</surname> <given-names>G.</given-names></name> <name><surname>Byrne</surname> <given-names>B.</given-names></name> <name><surname>de Gispert</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;From rewriting to remembering: common ground for conversational QA models,&#x0201D;</article-title> in <source>Proceedings of the 4th Workshop on NLP for Conversational AI</source>, eds. B. Liu, A. Papangelis, S. Ultes, A. Rastogi, Y.-N. Chen, G. Spithourakis, et al. (Dublin: Association for Computational Linguistics), <fpage>70</fpage>&#x02013;<lpage>76</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deutsch</surname> <given-names>D.</given-names></name> <name><surname>Bedrax-Weiss</surname> <given-names>T.</given-names></name> <name><surname>Roth</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Towards question-answering as an automatic metric for evaluating the content quality of a summary</article-title>. <source>Transact. Assoc. Comp. Linguist</source>. <volume>9</volume>, <fpage>774</fpage>&#x02013;<lpage>789</lpage>. <pub-id pub-id-type="doi">10.1162/tacl_a_00397</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dey</surname> <given-names>I.</given-names></name> <name><surname>Puntambekar</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Gengler</surname> <given-names>D.</given-names></name> <name><surname>Dickler</surname> <given-names>R.</given-names></name> <name><surname>Hirshfield</surname> <given-names>L. M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;The NICE framework: analyzing students&#x00027; nonverbal interactions during collaborative learning,&#x0201D;</article-title> in <source>Pre-conference Workshop on Collaboration Analytics at 13th International Learning Analytics and Knowledge Conference (LAK 2023)</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Society for Learning Analytics Research</publisher-name>).</citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dice</surname> <given-names>L. R.</given-names></name></person-group> (<year>1945</year>). <article-title>Measures of the amount of ecologic association between species</article-title>. <source>Ecology</source> <volume>26</volume>, <fpage>297</fpage>&#x02013;<lpage>302</lpage>. <pub-id pub-id-type="doi">10.2307/1932409</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dillenbourg</surname> <given-names>P.</given-names></name> <name><surname>Traum</surname> <given-names>D.</given-names></name></person-group> (<year>2006</year>). <article-title>Sharing solutions: persistence and grounding in multimodal collaborative problem solving</article-title>. <source>J. Learn. Sci</source>. <volume>15</volume>, <fpage>121</fpage>&#x02013;<lpage>151</lpage>. <pub-id pub-id-type="doi">10.1207/s15327809jls1501_9</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Donatelli</surname> <given-names>L.</given-names></name> <name><surname>Lai</surname> <given-names>K.</given-names></name> <name><surname>Brutti</surname> <given-names>R.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Towards situated AMR: creating a corpus of gesture AMR,&#x0201D;</article-title> in <source>Digital Human Modeling and Applications in Health, Safety, Ergonomics and Risk Management. Health, Operations Management, and Design</source>, ed. V. G. Duffy (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>93</fpage>&#x02013;<lpage>312</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eckert</surname> <given-names>M.</given-names></name> <name><surname>Strube</surname> <given-names>M.</given-names></name></person-group> (<year>2000</year>). <article-title>Dialogue acts, synchronizing units, and anaphora resolution</article-title>. <source>J. Semant</source>. <volume>17</volume>, <fpage>51</fpage>&#x02013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1093/jos/17.1.51</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eisenstein</surname> <given-names>J.</given-names></name> <name><surname>Andor</surname> <given-names>D.</given-names></name> <name><surname>Bohnet</surname> <given-names>B.</given-names></name> <name><surname>Collins</surname> <given-names>M.</given-names></name> <name><surname>Mimno</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Honest students from untrusted teachers: learning an interpretable question-answering pipeline from a pretrained language model</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2210.02498</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elazar</surname> <given-names>Y.</given-names></name> <name><surname>Basmov</surname> <given-names>V.</given-names></name> <name><surname>Goldberg</surname> <given-names>Y.</given-names></name> <name><surname>Tsarfaty</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>Text-based np enrichment</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.1162/tacl_a_00488</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Emami</surname> <given-names>A.</given-names></name> <name><surname>De La Cruz</surname> <given-names>N.</given-names></name> <name><surname>Trischler</surname> <given-names>A.</given-names></name> <name><surname>Suleman</surname> <given-names>K.</given-names></name> <name><surname>Cheung</surname> <given-names>J. C. K.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;A knowledge hunting framework for common sense reasoning,&#x0201D;</article-title> in <source>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</source>, eds. E. Riloff, D. Chiang, J. Hockenmaier, and J. Tsujii (Brussels: Association for Computational Linguistics), <fpage>1949</fpage>&#x02013;<lpage>1958</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eyal</surname> <given-names>M.</given-names></name> <name><surname>Baumel</surname> <given-names>T.</given-names></name> <name><surname>Elhadad</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Question answering as an automatic evaluation metric for news article summarization,&#x0201D;</article-title> in <source>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)</source>, ed. J. Burstein, C. Doran, and T. Solorio (Minneapolis, MN: Association for Computational Linguistics), <fpage>3938</fpage>&#x02013;<lpage>3948</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fischer</surname> <given-names>K.</given-names></name></person-group> (<year>2011</year>). <article-title>How people talk with robots: designing dialog to reduce user uncertainty</article-title>. <source>AI Mag</source>. <volume>32</volume>, <fpage>31</fpage>&#x02013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.1609/aimag.v32i4.2377</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ginzburg</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <source>The Interactive Stance: Meaning for Conversation</source>. <publisher-loc>Oxford</publisher-loc>: <publisher-name>Oxford University Press</publisher-name>.</citation>
</ref>
<ref id="B35">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Goldman</surname> <given-names>N. M.</given-names></name></person-group> (<year>1977</year>). <source>Sentence Paraphrasing From a Conceptual Base</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Communications of the ACM</publisher-name>, <fpage>481</fpage>&#x02013;<lpage>507</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gong</surname> <given-names>T.</given-names></name> <name><surname>Lyu</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Zheng</surname> <given-names>M.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Multimodal-GPT: a vision and language model for dialogue with humans</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2305.04790</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gunasekara</surname> <given-names>C.</given-names></name> <name><surname>Feigenblat</surname> <given-names>G.</given-names></name> <name><surname>Sznajder</surname> <given-names>B.</given-names></name> <name><surname>Aharonov</surname> <given-names>R.</given-names></name> <name><surname>Joshi</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Using question answering rewards to improve abstractive summarization,&#x0201D;</article-title> in <source>Findings of the Association for Computational Linguistics: EMNLP 2021</source>, eds. M.-F. Moens, X. Huang, L. Specia, and S. W.-t. Yih (Punta Cana: Association for Computational Linguistics), <fpage>518</fpage>&#x02013;<lpage>526</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hadley</surname> <given-names>L. V.</given-names></name> <name><surname>Naylor</surname> <given-names>G.</given-names></name> <name><surname>Hamilton</surname> <given-names>A. F. C.</given-names></name></person-group> (<year>2022</year>). <article-title>A review of theories and methods in the science of face-to-face social interaction</article-title>. <source>Nat. Rev. Psychol</source>. <volume>1</volume>, <fpage>42</fpage>&#x02013;<lpage>54</lpage>. <pub-id pub-id-type="doi">10.1038/s44159-021-00008-w</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jacqmin</surname> <given-names>L.</given-names></name> <name><surname>Rojas Barahona</surname> <given-names>L. M.</given-names></name> <name><surname>Favre</surname> <given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;&#x0201C;do you follow me?&#x0201D;: a survey of recent approaches in dialogue state tracking,&#x0201D;</article-title> in <source>Proceedings of the 23rd Annual Meeting of the Special Interest Group on Discourse and Dialogue</source>, eds. O. Lemon, D. Hakkani-Tur, J. J. Li, A. Ashrafzadeh, D. H. Garcia, M. Alikhani, et al. (<publisher-loc>Edinburgh</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>336</fpage>&#x02013;<lpage>350</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Katz</surname> <given-names>U.</given-names></name> <name><surname>Geva</surname> <given-names>M.</given-names></name> <name><surname>Berant</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Inferring implicit relations in complex questions with language models</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.18653/v1/2022.findings-emnlp.188</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khebour</surname> <given-names>I.</given-names></name> <name><surname>Brutti</surname> <given-names>R.</given-names></name> <name><surname>Dey</surname> <given-names>I.</given-names></name> <name><surname>Dickler</surname> <given-names>R.</given-names></name> <name><surname>Sikes</surname> <given-names>K.</given-names></name> <name><surname>Lai</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>The weights task dataset: a multimodal dataset of collaboration in a situated task</article-title>. <source>J. Open Human. Data</source>. 10. <pub-id pub-id-type="doi">10.5334/johd.168</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Khebour</surname> <given-names>I. K.</given-names></name> <name><surname>Lai</surname> <given-names>K.</given-names></name> <name><surname>Bradford</surname> <given-names>M.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Brutti</surname> <given-names>R. A.</given-names></name> <name><surname>Tam</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2024b</year>). <article-title>&#x0201C;Common ground tracking in multimodal dialogue,&#x0201D;</article-title> in <source>Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)</source> (<publisher-loc>Torino</publisher-loc>: <publisher-name>ELRA and ICCL</publisher-name>), <fpage>3587</fpage>&#x02013;<lpage>3602</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khosla</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <name><surname>Manuvinakurike</surname> <given-names>R.</given-names></name> <name><surname>Ng</surname> <given-names>V.</given-names></name> <name><surname>Poesio</surname> <given-names>M.</given-names></name> <name><surname>Strube</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;The CODI-CRAC 2021 shared task on anaphora, bridging, and discourse deixis in dialogue,&#x0201D;</article-title> in <source>Proceedings of the CODI-CRAC 2021 Shared Task on Anaphora, Bridging, and Discourse Deixis in Dialogue</source>, eds. S. Khosla, R. Manuvinakurike, V. Ng, M. Poesio, M., Strube, and C. Ros&#x000E9; (Punta Cana: Association for Computational Linguistics), <fpage>1</fpage>&#x02013;<lpage>15</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krishnaswamy</surname> <given-names>N.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Generating a novel dataset of multimodal referring expressions,&#x0201D;</article-title> in <source>Proceedings of the 13th International Conference on Computational Semantics</source> - <italic>Short Papers</italic>, eds. S. Dobnik, S. Chatzikyriakidis, and V. Demberg (Gothenburg: Association for Computational Linguistics), <fpage>44</fpage>&#x02013;<lpage>51</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kruijff</surname> <given-names>G.-J. M.</given-names></name> <name><surname>Lison</surname> <given-names>P.</given-names></name> <name><surname>Benjamin</surname> <given-names>T.</given-names></name> <name><surname>Jacobsson</surname> <given-names>H.</given-names></name> <name><surname>Zender</surname> <given-names>H.</given-names></name> <name><surname>Kruijff-Korbayov&#x000E1;</surname> <given-names>I.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>&#x0201C;Situated dialogue processing for human-robot interaction,&#x0201D;</article-title> in <source>Cognitive Systems</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>311</fpage>&#x02013;<lpage>364</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>Talukdar</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;NILE: natural language inference with faithful natural language explanations,&#x0201D;</article-title> in <source>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</source>, eds. D. Jurafsky, J. Chai, N. Schluter, and J. Tetreault (Stroudsburg, PA: Association for Computational Linguistics), <fpage>8730</fpage>&#x02013;<lpage>8742</lpage>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>L.</given-names></name> <name><surname>Long</surname> <given-names>L. H.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Lei</surname> <given-names>W.</given-names></name> <name><surname>Chua</surname> <given-names>T.-S.</given-names></name></person-group> (<year>2021</year>). <article-title>Dialogue state tracking with incremental reasoning</article-title>. <source>Transact. Assoc. Comp. Linguist</source>. <volume>9</volume>, <fpage>557</fpage>&#x02013;<lpage>569</lpage>. <pub-id pub-id-type="doi">10.1162/tacl_a_00384</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>L.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Hong</surname> <given-names>R.</given-names></name> <name><surname>Chua</surname> <given-names>T.-S.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Knowledge-aware multimodal dialogue systems,&#x0201D;</article-title> in <source>Proceedings of the 26th ACM International Conference on Multimedia</source> (<publisher-loc>New York, NY</publisher-loc>), <fpage>801</fpage>&#x02013;<lpage>809</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>B. Y.</given-names></name> <name><surname>Lee</surname> <given-names>S.</given-names></name> <name><surname>Qiao</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Common sense beyond English: evaluating and improving multilingual language models for commonsense reasoning,&#x0201D;</article-title> in <source>em Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)</source>, eds. C. Zong, F. Xia, W. Li, and R. Navigli (Stroudsburg, PA: Association for Computational Linguistics), <fpage>1274</fpage>&#x02013;<lpage>1287</lpage>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mao</surname> <given-names>H. H.</given-names></name> <name><surname>Majumder</surname> <given-names>B. P.</given-names></name> <name><surname>McAuley</surname> <given-names>J.</given-names></name> <name><surname>Cottrell</surname> <given-names>G.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Improving neural story generation by targeted common sense grounding,&#x0201D;</article-title> in <source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</source>, eds. K. Inui, J. Jiang, V. Ng, and X. Wan (Hong Kong: Association for Computational Linguistics), <fpage>988</fpage>&#x02013;<lpage>5993</lpage>.</citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McKeown</surname> <given-names>K.</given-names></name></person-group> (<year>1983</year>). <article-title>Paraphrasing questions using given and new information</article-title>. <source>Am. J. Comp. Linguist</source>. <volume>9</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Mekala</surname> <given-names>D.</given-names></name> <name><surname>Vu</surname> <given-names>T.</given-names></name> <name><surname>Schick</surname> <given-names>T.</given-names></name> <name><surname>Shang</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Leveraging QA datasets to improve generative data augmentation,&#x0201D;</article-title> in <source>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</source>, eds. Y. Goldberg, Z. Kozareva, and Y. Zhang (<publisher-loc>Abu Dhabi</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>9737</fpage>&#x02013;<lpage>9750</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="thesis"><person-group person-group-type="author"><name><surname>M&#x000FC;ller</surname> <given-names>M.-C.</given-names></name></person-group> (<year>2008</year>). <source>Fully Automatic Resolution of &#x02018;it&#x00027;</source>, &#x02018;<italic>This&#x00027;, and &#x02018;that in Unrestricted Multi-Party Dialog</italic> (PhD thesis). <publisher-name>Universit&#x000E4;t T&#x000FC;bingen</publisher-name>, <publisher-loc>T&#x000FC;bingen, Germany</publisher-loc>.</citation>
</ref>
<ref id="B54">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Muraki</surname> <given-names>K.</given-names></name></person-group> (<year>1982</year>). <article-title>&#x0201C;On a semantic model for multi-lingual paraphrasing,&#x0201D;</article-title> in <source>Coling 1982: Proceedings of the Ninth International Conference on Computational Linguistics</source> (<publisher-loc>Stroudsburg, PA</publisher-loc>).</citation>
</ref>
<ref id="B55">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Nishida</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Envisioning conversation: toward understanding and augmenting common ground,&#x0201D;</article-title> in <source>Proceedings of the 2018 International Conference on Advanced Visual Interfaces, AVI &#x00027;18</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>).</citation>
</ref>
<ref id="B56">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Obiso</surname> <given-names>T.</given-names></name> <name><surname>Ye</surname> <given-names>B.</given-names></name> <name><surname>Rim</surname> <given-names>K.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Semantically enriched text generation for QA through dense paraphrasing,&#x0201D;</article-title> in <source>Proceedings of the 7th International Conference on Natural Language and Speech Processing (ICNLSP 2024)</source>, eds. M. Abbas and A. A. Freihat (Trento: Association for Computational Linguistics), <fpage>279</fpage>&#x02013;<lpage>286</lpage>. Available at: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2024.icnlsp-1.30">https://aclanthology.org/2024.icnlsp-1.30</ext-link></citation>
</ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ohmer</surname> <given-names>X.</given-names></name> <name><surname>Duda</surname> <given-names>M.</given-names></name> <name><surname>Bruni</surname> <given-names>E.</given-names></name></person-group> (<year>2022</year>). <article-title>Emergence of hierarchical reference systems in multi-agent communication</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2203.13176</pub-id></citation>
</ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><collab>OpenAI</collab> <name><surname>Achiam</surname> <given-names>J.</given-names></name> <name><surname>Adler</surname> <given-names>S.</given-names></name> <name><surname>Agarwal</surname> <given-names>S.</given-names></name> <name><surname>Ahmad</surname> <given-names>L.</given-names></name> <name><surname>Akkaya</surname> <given-names>I.</given-names></name> <etal/></person-group>. (<year>2024</year>). <source>Gpt-4 Technical Report</source>.</citation>
</ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><collab>OpenAI</collab></person-group> (<year>2023</year>). <source>Gpt-4 Technical Report</source>.</citation>
</ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ou</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>C.</given-names></name> <name><surname>Tang</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>F.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;DialogBench: evaluating LLMs as human-like dialogue systems,&#x0201D;</article-title> in <source>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)</source>, eds. K. Duh, H. Gomez, and S. Bethard (Mexico City: Association for Computational Linguistics), <fpage>6137</fpage>&#x02013;<lpage>6170</lpage>.</citation>
</ref>
<ref id="B61">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Pacuit</surname> <given-names>E.</given-names></name></person-group> (<year>2017</year>). <source>Neighborhood Semantics for Modal Logic</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parikh</surname> <given-names>A.</given-names></name> <name><surname>T&#x000E4;ckstr&#x000F6;m</surname> <given-names>O.</given-names></name> <name><surname>Das</surname> <given-names>D.</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;A decomposable attention model for natural language inference,&#x0201D;</article-title> in <source>Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing</source>, eds. J. Su, K. Duh, and X. Carreras (Austin, TX: Association for Computational Linguistics), <fpage>2249</fpage>&#x02013;<lpage>2255</lpage>. <pub-id pub-id-type="pmid">22759459</pub-id></citation></ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Poria</surname> <given-names>S.</given-names></name> <name><surname>Gelbukh</surname> <given-names>A.</given-names></name> <name><surname>Cambria</surname> <given-names>E.</given-names></name> <name><surname>Hussain</surname> <given-names>A.</given-names></name> <name><surname>Huang</surname> <given-names>G.-B.</given-names></name></person-group> (<year>2014</year>). <article-title>Emosenticspace: a novel framework for affective common-sense reasoning</article-title>. <source>Knowl. Based Syst</source>. <volume>69</volume>:<fpage>108</fpage>&#x02013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2014.06.011</pub-id></citation>
</ref>
<ref id="B64">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>1995</year>). <source>The Generative Lexicon</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>.</citation>
</ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pustejovsky</surname> <given-names>J.</given-names></name> <name><surname>Krishnaswamy</surname> <given-names>N.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;VoxML: A visualization modeling language,&#x0201D;</article-title> in <source>Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC&#x00027;16)</source>, eds. N. Calzolari, K. Choukri, T. Declerck, S. Goggi, M. Grobelnik, B. Maegaard, et al. [Portoro&#x0017E;: European Language Resources Association (ELRA)], <fpage>4606</fpage>&#x02013;<lpage>4613</lpage>.</citation>
</ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rim</surname> <given-names>K.</given-names></name> <name><surname>Tu</surname> <given-names>J.</given-names></name> <name><surname>Ye</surname> <given-names>B.</given-names></name> <name><surname>Verhagen</surname> <given-names>M.</given-names></name> <name><surname>Holderness</surname> <given-names>E.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;The coreference under transformation labeling dataset: entity tracking in procedural texts using event models,&#x0201D;</article-title> in <source>Findings of the Association for Computational Linguistics: ACL 2023</source>, eds. A. Rogers, J. Boyd-Graber, and N. Okazaki (Toronto, ON: Association for Computational Linguistics), <fpage>12448</fpage>&#x02013;<lpage>12460</lpage>.</citation>
</ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saha</surname> <given-names>A.</given-names></name> <name><surname>Khapra</surname> <given-names>M.</given-names></name> <name><surname>Sankaranarayanan</surname> <given-names>K.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Towards building large scale multimodal domain-aware conversation systems,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence, Vol</source>. 32 (Washington, DC).</citation>
</ref>
<ref id="B68">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Scheutz</surname> <given-names>M.</given-names></name> <name><surname>Cantrell</surname> <given-names>R.</given-names></name> <name><surname>Schermerhorn</surname> <given-names>P.</given-names></name></person-group> (<year>2011</year>). <article-title>Toward humanlike task-based dialogue processing for human robot interaction</article-title>. <source>Ai Mag</source>. <volume>32</volume>, <fpage>77</fpage>&#x02013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1609/aimag.v32i4.2381</pub-id></citation>
</ref>
<ref id="B69">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schick</surname> <given-names>T.</given-names></name> <name><surname>Sch&#x000FC;tze</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Exploiting cloze-questions for few-shot text classification and natural language inference,&#x0201D;</article-title> in <source>Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume</source>, eds. P. Merlo, J. Tiedemann, and R. Tsarfaty (Stroudsburg, PA: Association for Computational Linguistics), <fpage>255</fpage>&#x02013;<lpage>269</lpage>.</citation>
</ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>S&#x000F8;rensen</surname> <given-names>T.</given-names></name></person-group> (<year>1948</year>). <article-title>A method of establishing groups of equal amplitude in plant sociology based on similarity of species content and its application to analyses of the vegetation on danish commons</article-title>. <source>Biologiske Skrifter</source> <volume>5</volume>, <fpage>1</fpage>&#x02013;<lpage>34</lpage>.</citation>
</ref>
<ref id="B71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>C.</given-names></name> <name><surname>Shute</surname> <given-names>V. J.</given-names></name> <name><surname>Stewart</surname> <given-names>A.</given-names></name> <name><surname>Yonehiro</surname> <given-names>J.</given-names></name> <name><surname>Duran</surname> <given-names>N.</given-names></name> <name><surname>D&#x00027;Mello</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Towards a generalized competency model of collaborative problem solving</article-title>. <source>Comput. Educ</source>. <volume>143</volume>:<fpage>103672</fpage>. <pub-id pub-id-type="doi">10.1016/j.compedu.2019.103672</pub-id></citation>
</ref>
<ref id="B72">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sur&#x000ED;s</surname> <given-names>D.</given-names></name> <name><surname>Duarte</surname> <given-names>A.</given-names></name> <name><surname>Salvador</surname> <given-names>A.</given-names></name> <name><surname>Torres</surname> <given-names>J.</given-names></name> <name><surname>Gir&#x000F3;-i Nieto</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Cross-modal embeddings for video and audio retrieval,&#x0201D;</article-title> in <source>Proceedings of the European Conference on Computer Vision (ECCV) Workshops</source> (<publisher-loc>New York, NY</publisher-loc>).</citation>
</ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Martin</surname> <given-names>L.</given-names></name> <name><surname>Stone</surname> <given-names>K. R.</given-names></name> <name><surname>Albert</surname> <given-names>P.</given-names></name> <name><surname>Almahairi</surname> <given-names>A.</given-names></name> <name><surname>Babaei</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Llama 2: Open foundation and fine-tuned chat models</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2307.09288</pub-id></citation>
</ref>
<ref id="B74">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tran</surname> <given-names>V.-K.</given-names></name> <name><surname>Nguyen</surname> <given-names>L.-M.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Semantic refinement gru-based neural language generation for spoken dialogue systems,&#x0201D;</article-title> in <source>Computational Linguistics: 15th International Conference of the Pacific Association for Computational Linguistics, PACLING 2017, Yangon, Myanmar, August 16-18, 2017, Revised Selected Papers 15</source> (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>63</fpage>&#x02013;<lpage>75</lpage>.</citation>
</ref>
<ref id="B75">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Traum</surname> <given-names>D.</given-names></name></person-group> (<year>1994</year>). <source>A Computational Theory of Grounding in Natural Language Conversation</source>. <publisher-loc>Rochester, NY</publisher-loc>: <publisher-name>University of Rochester</publisher-name>. <pub-id pub-id-type="pmid">34242331</pub-id></citation></ref>
<ref id="B76">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>J.</given-names></name> <name><surname>Holderness</surname> <given-names>E.</given-names></name> <name><surname>Maru</surname> <given-names>M.</given-names></name> <name><surname>Conia</surname> <given-names>S.</given-names></name> <name><surname>Rim</surname> <given-names>K.</given-names></name> <name><surname>Lynch</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2022a</year>). <article-title>&#x0201C;SemEval-2022 task 9: R2VQ - competence-based multimodal question answering,&#x0201D;</article-title> in <source>Proceedings of the 16th International Workshop on Semantic Evaluation (SemEval-2022)</source>, eds. G. Emerson, N. Schluter, G. Stanovsky, R. Kumar, A., Palmer, N. Schneider (Seattle, WA: Association for Computational Linguistics), <fpage>1244</fpage>&#x02013;<lpage>1255</lpage>.</citation>
</ref>
<ref id="B77">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>J.</given-names></name> <name><surname>Obiso</surname> <given-names>T.</given-names></name> <name><surname>Ye</surname> <given-names>B.</given-names></name> <name><surname>Rim</surname> <given-names>K.</given-names></name> <name><surname>Xu</surname> <given-names>K.</given-names></name> <name><surname>Yue</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;GLAMR: augmenting AMR with GL-VerbNet event structure,&#x0201D;</article-title> in <source>Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)</source>, eds. N. Calzolari, M.-Y. Kan, V. Hoste, A. Lenci., S. Sakti, and N. Xue (Torino: ELRA and ICCL), <fpage>7746</fpage>&#x02013;<lpage>7759</lpage>.</citation>
</ref>
<ref id="B78">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>J.</given-names></name> <name><surname>Rim</surname> <given-names>K.</given-names></name> <name><surname>Holderness</surname> <given-names>E.</given-names></name> <name><surname>Ye</surname> <given-names>B.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Dense paraphrasing for textual enrichment,&#x0201D;</article-title> in <source>Proceedings of the 15th International Conference on Computational Semantics</source>, eds. M. Amblard, and E. Breitholtz (Nancy: Association for Computational Linguistics), <fpage>39</fpage>&#x02013;<lpage>49</lpage>.</citation>
</ref>
<ref id="B79">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>J.</given-names></name> <name><surname>Rim</surname> <given-names>K.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2022b</year>). <article-title>&#x0201C;Competence-based question generation,&#x0201D;</article-title> in <source>Proceedings of the 29th International Conference on Computational Linguistics</source>, eds. N. Calzolari, C.-R. Huang, H. Kim, J. Pustejovsky, L. Wanner, K.-S. Choi, et al. (Gyeongju: International Committee on Computational Linguistics), <fpage>1521</fpage>&#x02013;<lpage>1533</lpage>.</citation>
</ref>
<ref id="B80">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>van Benthem</surname> <given-names>J.</given-names></name> <name><surname>Fern&#x000E1;ndez-Duque</surname> <given-names>D.</given-names></name> <name><surname>Pacuit</surname> <given-names>E.</given-names></name></person-group> (<year>2014</year>). <article-title>Evidence and plausibility in neighborhood structures</article-title>. <source>Ann. Pure Appl. Logic</source> <volume>165</volume>, <fpage>106</fpage>&#x02013;<lpage>133</lpage>. <pub-id pub-id-type="doi">10.1016/j.apal.2013.07.007</pub-id></citation>
</ref>
<ref id="B81">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vinyals</surname> <given-names>O.</given-names></name> <name><surname>Le</surname> <given-names>Q.</given-names></name></person-group> (<year>2015</year>). <article-title>A neural conversational model</article-title>. <source>arXiv</source> [preprint].</citation>
</ref>
<ref id="B82">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Z.</given-names></name> <name><surname>Luan</surname> <given-names>Y.</given-names></name> <name><surname>Rashkin</surname> <given-names>H.</given-names></name> <name><surname>Reitter</surname> <given-names>D.</given-names></name> <name><surname>Tomar</surname> <given-names>G. S.</given-names></name></person-group> (<year>2021</year>). <article-title>Conqrr: Conversational query rewriting for retrieval with reinforcement learning</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.18653/v1/2022.emnlp-main.679</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B83">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Ye</surname> <given-names>B.</given-names></name> <name><surname>Tu</surname> <given-names>J.</given-names></name> <name><surname>Jezek</surname> <given-names>E.</given-names></name> <name><surname>Pustejovsky</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Interpreting logical metonymy through dense paraphrasing,&#x0201D;</article-title> in <source>Proceedings of the 44th Annual Meeting of the Cognitive Science Society, CogSci 2022</source>, eds. J. Culbertson, H. Rabagliati, V. C. Ramenzoni, and A. Perfors (<publisher-loc>Toronto, ON</publisher-loc>). Available at: <ext-link ext-link-type="uri" xlink:href="http://cognitivesciencesociety.org">http://cognitivesciencesociety.org</ext-link></citation>
</ref>
<ref id="B84">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ye</surname> <given-names>F.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Stern</surname> <given-names>S.</given-names></name> <name><surname>Yilmaz</surname> <given-names>E.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;MetaASSIST: robust dialogue state tracking with meta learning,&#x0201D;</article-title> in <source>Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing</source>, eds. Y. Goldberg, Z. Kozareva, and Y. Zhang (Abu Dhabi: Association for Computational Linguistics), <fpage>1157</fpage>&#x02013;<lpage>1169</lpage>.</citation>
</ref>
<ref id="B85">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yi</surname> <given-names>Z.</given-names></name> <name><surname>Ouyang</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Liao</surname> <given-names>T.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Shen</surname> <given-names>Y.</given-names></name></person-group> (<year>2024</year>). <article-title>A survey on recent advances in llm-based multi-turn dialogue systems</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2402.18013</pub-id></citation>
</ref>
<ref id="B86">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>S.</given-names></name> <name><surname>Fu</surname> <given-names>C.</given-names></name> <name><surname>Zhao</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Sun</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>A survey on multimodal large language models</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.1093/nsr/nwae403</pub-id><pub-id pub-id-type="pmid">39679213</pub-id></citation></ref>
<ref id="B87">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhai</surname> <given-names>W.</given-names></name> <name><surname>Feng</surname> <given-names>M.</given-names></name> <name><surname>Zubiaga</surname> <given-names>A.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;HIT&#x00026;QMUL at SemEval-2022 task 9: Label-enclosed generative question answering (LEG-QA),&#x0201D;</article-title> in <source>Proceedings of the 16th International Workshop on Semantic Evaluation (SemEval-2022)</source>, eds. G. Emerson, N. Schluter, G. Stanovsky, R. Kumar, A. Palmer, N. Schneider (Seattle, WA: Association for Computational Linguistics), <fpage>1256</fpage>&#x02013;<lpage>1262</lpage>.</citation>
</ref>
<ref id="B88">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Dugan</surname> <given-names>L.</given-names></name> <name><surname>Xu</surname> <given-names>H.</given-names></name> <name><surname>Callison-burch</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Exploring the curious case of code prompts,&#x0201D;</article-title> in <source>Proceedings of the 1st Workshop on Natural Language Reasoning and Structured Explanations (NLRSE)</source>, eds. B. Dalvi Mishra, G. Durrett, P. Jansen, D. Neves Ribeiro, and J. Wei (Toronto, ON: Association for Computational Linguistics), <fpage>9</fpage>&#x02013;<lpage>17</lpage>.</citation>
</ref>
<ref id="B89">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>R.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Jiao</surname> <given-names>F.</given-names></name> <name><surname>Do</surname> <given-names>X. L.</given-names></name> <name><surname>Qin</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;Retrieving multimodal information for augmented generation: a survey,&#x0201D;</article-title> in <source>Findings of the Association for Computational Linguistics: EMNLP 2023</source>, eds. H. Bouamor, J. Pino, and K. Bali (Singapore: Association for Computational Linguistics), <fpage>4736</fpage>&#x02013;<lpage>4756</lpage>.</citation>
</ref>
<ref id="B90">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>H.</given-names></name> <name><surname>Huang</surname> <given-names>M.</given-names></name> <name><surname>Zhu</surname> <given-names>X.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Context-aware natural language generation for spoken dialogue systems,&#x0201D;</article-title> in <source>Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: Technical Papers, 2032-2041</source>, eds. Y. Matsumoto, and R. Prasad (Osaka: The COLING 2016 Organizing Committee).</citation>
</ref>
</ref-list>
</back>
</article>