<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1512099</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1512099</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A relevance model of human sparse communication in cooperation</article-title>
<alt-title alt-title-type="left-running-head">Jiang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1512099">10.3389/frobt.2025.1512099</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Jiang</surname>
<given-names>Kaiwen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2752149/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Boxuan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2870915/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sadaghdar</surname>
<given-names>Anahita</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Limb</surname>
<given-names>Rebekah</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Gao</surname>
<given-names>Tao</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Statistics and Probability</institution>, <institution>Michigan State University</institution>, <addr-line>East Lansing</addr-line>, <addr-line>MI</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Statistics and Data Science</institution>, <institution>University of California Los Angeles</institution>, <addr-line>Los Angeles</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Psychology</institution>, <institution>University of California Los Angeles</institution>, <addr-line>Los Angeles</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Communication</institution>, <institution>University of California Los Angeles</institution>, <addr-line>Los Angeles</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2669398/overview">Ning Wang</ext-link>, University of Southern California, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2348432/overview">Paulina Tsvetkova</ext-link>, Bulgarian Academy of Sciences, Bulgaria</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/724331/overview">Caroline Ponzoni Carvalho Chanel</ext-link>, Institut Sup&#xe9;rieur de l&#x27;A&#xe9;ronautique et de l&#x27;Espace (ISAE-SUPAERO), France</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Kaiwen Jiang, <email>jiangk11@msu.edu</email>; Tao Gao, <email>taogao@ucla.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>30</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1512099</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Jiang, Jiang, Sadaghdar, Limb and Gao.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Jiang, Jiang, Sadaghdar, Limb and Gao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Human real-time communication creates a limitation on the flow of information, which requires the transfer of carefully chosen and condensed data in various situations. We introduce a model that explains how humans choose information for communication by utilizing the concept of &#x201c;relevance&#x201d; derived from decision-making theory and Theory of Mind (ToM). We evaluated the model by conducting experiments where human participants and an artificial intelligence (AI) agent assist each other to avoid multiple traps in a simulated navigation task. The relevance model accurately depicts how humans choose which trap to communicate. It also outperforms GPT-4, which participates in the same task by responding to prompts that describe the game settings and rules. Furthermore, we demonstrated that when humans received assisting information from an AI agent, they achieved a much higher performance and gave higher ratings to the AI when it utilized the relevance model compared to a heuristic model. Together, these findings provide compelling evidence that a relevance model rooted in decision theory and ToM can effectively capture the sparse and spontaneous nature of human communication.</p>
</abstract>
<kwd-group>
<kwd>relevance</kwd>
<kwd>decision theory</kwd>
<kwd>theory of mind</kwd>
<kwd>POMDP</kwd>
<kwd>large language model</kwd>
<kwd>artificial intelligence</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Intelligence in Robotics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>In a football game, the receiver catches the ball. Despite three opponent defenders rapidly approaching him, he accelerates without noticing them. Within that split second, you can only alert him about one of the opponents. How do you make that choice in fast and instantaneous communication?</p>
<p>Living in a fast-paced modern world, we constantly face an overload of information while needing to make quick decisions of what to communicate. This necessity for rapid decision-making applies to contexts as diverse as emergency room operations, stock trading, and professional kitchens. In these domains, communication must be swift and effective, conveying substantial information efficiently.</p>
<p>But how do humans manage to distill complex information into concise, impromptu communicative signals? How do we determine what information to share? In this paper, we use pointing&#x2014;a remarkably simple form of communication&#x2014;as a case study to investigate how humans swiftly and effectively choose which information to communicate.</p>
<p>Pointing has been extensively studied within the fields of cognitive science and developmental psychology. Notably in developmental psychology, humans develop pointing even prior to the acquisition of language. For instance, infants have been shown to understand the implications of pointing from contextual cues and use pointing at 12 months old (<xref ref-type="bibr" rid="B51">Tomasello, 2010</xref>). Toddlers aged 18&#x2013;36 months even point in consideration of others&#x2019; perspectives, exhibiting greater amounts of pointing when their partner&#x2019;s view is obstructed (<xref ref-type="bibr" rid="B6">Franco and Gagliano, 2001</xref>).</p>
<p>Additionally, <xref ref-type="bibr" rid="B29">Misyak et al. (2016)</xref> studies human adult pointing behavior to demonstrate that human pointing is extremely overloaded and indirect. The same signal can have multiple different meanings and these meanings can communicate something beyond the object&#x2019;s visual features, such as what to do with the referent and why the pointer employed such a gesture.</p>
<p>Pointing is overloaded, indirect, and sparse. How can humans accurately interpret it in a mere instant? This question may be answered by the interaction of humans and their surrounding environment: how we act in the environment imposes a critical constraint on the interpretation of pointing signals. From the perspective of decision theory, if we take upon the widely accepted assumption of human rationality, it is imperative for one to perform pointing signals that facilitate actions which lead to achieving the highest utility in the most efficient manner possible. For example, when a chef points to the cookware cabinet, the sous chef may not know what cookware the chef is asking for. However, in the context of making a salad, it is more likely that the chef is asking for a knife, while in the context of making a souffle, the chef is more likely to ask for a mixer. The meaning of the same pointing gesture can be dynamically determined by the task involving the sender and interpreter. Therefore, the richness and flexibility does not arise from the gesture itself, but from the variety of contextual constraints induced by tasks. In turn, capturing the decision making aspect of the task, estimating the utility of what has been done and what could be done, is the key to deciphering pointing.</p>
<p>What bridges the communication and contexts is a concept called relevance, which is widely studied in the field of linguistics. <xref ref-type="bibr" rid="B11">Grice (1975)</xref> introduced the maxim of relevance, asserting that speakers must provide information that is relevant to the current conversation if they wish for listeners to correctly understand their meaning. Similarly, listeners can successfully understand the intended meaning when they assume the speaker is adhering to this maxim. In Grice&#x2019;s definition, a sentence is relevant if its meaning continues the dialogue. For example, if one asks where the scissors are and receives the response, &#x201c;I cut paper in my room.&#x201d; Assuming this response is relevant, it must continues the conversation even though it does not mention scissors at all. To afford the continuation of the conversation, the response must mean I cut paper with the scissors in my room, so we may find the scissors in the respondent&#x2019;s room. <xref ref-type="bibr" rid="B55">Wilson and Sperber (2006)</xref> added that relevance can be assessed in terms of cognitive effect and processing effort. In their definition of relevance, relevant information helps the listener by creating a worthwhile difference in their understanding of the world. Notably so far, the discussions of relevance in linguistics focus primarily on the sentences in conversations. We aim to extend their work to non-verbal overloaded communication by incorporating components like Theory of Mind (ToM) and task-planning. We will define a sequential decision-making task with two agents: one helper who communicates with one actor only through pointing.</p>
<p>Recently, decision-theory is increasingly incorporated into communication studies. The rational speech act (RSA) model (<xref ref-type="bibr" rid="B7">Frank and Goodman, 2012</xref>), for instance, is a linguistic model developed from reference games, in which one attempts to find a target among distractors based on a description. The RSA model assigns each utterance a utility based on how much it can direct the listener&#x2019;s belief towards the true target. Extending this reference game to decision-making, relevance emerges as an utterance&#x2019;s ability to increase the listener&#x2019;s reward. Studies on pedagogy reveal that individuals meticulously select their instructions by considering its impact on the listener&#x2019;s performance and belief (<xref ref-type="bibr" rid="B47">Sumers et al., 2023a</xref>; <xref ref-type="bibr" rid="B14">Ho et al., 2016</xref>). Another study shows the proficiency of robot assistants in providing the most helpful&#x2013;even if somewhat distorted&#x2013;observations to aid task-solving (<xref ref-type="bibr" rid="B38">Reddy et al., 2021</xref>). Another line of work demonstrates on basing signal generation and inference of extremely overloaded signals. Interpreting a helper&#x2019;s signals by assuming that the helper is trying to be maximally helpful significantly improves agent&#x2019;s performance (<xref ref-type="bibr" rid="B17">Jiang et al., 2022</xref>). To further our understanding of relevance, we intend to build a relevance model-based decision theory and ToM that can capture the essence of human communication and support human-AI interaction.</p>
<p>The approach of modeling human decision-making through utility calculations has a long-standing tradition, with prospect theory being one notable example (<xref ref-type="bibr" rid="B5">de Souza et al., 2020</xref>). To further model how agents interact with their environment over time, sequential decision-making frameworks such as Markov decision process (MDP) are commonly used (<xref ref-type="bibr" rid="B19">Knox and Stone, 2012</xref>; <xref ref-type="bibr" rid="B25">Levin et al., 2002</xref>). When focusing on decisions driven by the agent&#x2019;s internal mental states, it&#x2019;s important to incorporate the agent&#x2019;s beliefs and perception of information. In contexts like human navigation (<xref ref-type="bibr" rid="B45">Stankiewicz et al., 2006</xref>) and inverse reinforcement learning (<xref ref-type="bibr" rid="B3">Baker et al., 2011</xref>), where agents typically have access to only partial information, the partially observable Markov decision process (POMDP) framework is frequently employed. POMDPs are well-suited for capturing both human and autonomous robot behavior (<xref ref-type="bibr" rid="B21">Kurniawati and Yadav, 2016</xref>), making them a valuable tool in human-robot interaction (<xref ref-type="bibr" rid="B56">You et al., 2023</xref>; <xref ref-type="bibr" rid="B42">Singh et al., 2022</xref>; <xref ref-type="bibr" rid="B30">Nikolaidis et al., 2017</xref>; <xref ref-type="bibr" rid="B54">Wang et al., 2016</xref>; <xref ref-type="bibr" rid="B31">Nikolaidis et al., 2015</xref>; <xref ref-type="bibr" rid="B22">Lam and Sastry, 2014</xref>; <xref ref-type="bibr" rid="B49">Taha et al., 2011</xref>), autonomous driving (<xref ref-type="bibr" rid="B37">Qiu et al., 2020</xref>), and communication between humans (<xref ref-type="bibr" rid="B14">Ho et al., 2016</xref>) or between human and robot (<xref ref-type="bibr" rid="B24">Le Guillou et al., 2023</xref>). As an example, <xref ref-type="bibr" rid="B56">You et al. (2023)</xref> designed a POMDP-based robotic system to predict the human collaborator&#x2019;s action and coordinate in a dual-agent task. Multi-agent POMDP models such as interactive POMDP (<xref ref-type="bibr" rid="B10">Gmytrasiewicz and Doshi, 2005</xref>) and decentralized POMDP (<xref ref-type="bibr" rid="B32">Oliehoek and Amato, 2016</xref>) further study the collaboration and communication between agents (<xref ref-type="bibr" rid="B9">Gmytrasiewicz and Adhikari, 2019</xref>).</p>
<p>Even though the POMDP model remarkably succeeded in modeling human decision-making in partially observable environments, it is not our main objective to build robotic systems to physically interact with humans and the environment, like in (<xref ref-type="bibr" rid="B56">You et al., 2023</xref>). Instead, we focus on modeling how humans select and interpret communicative signals, which only changes the mind of the agents. In the following section, we propose a utility-based relevance framework. With this objective, two crucial components, action prediction and action evaluation, depend on utility computed via POMDP in this study. Nevertheless, POMDP is not indispensable and may be substituted with alternative agent models capable of utility estimation. In our experiments, POMDP is adopted to model the decisions made by the player who physically changes the environment, but not to model the signal selection of the helper as they only change the player&#x2019;s mind. Also with our objective, we measure the consistency of the predictions by our model with human behavior, in addition to the joint reward gained by the collaboration between human and agent.</p>
</sec>
<sec id="s2">
<title>2 Model</title>
<p>We design a game inspired by the impromptu and sparse communication in the football example in the introduction. In this game, a player navigates the map from a starting point to a goal. We design 5 by 5 maps with examples shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. As we assume a small cost of 1 for each step coupled with a substantial reward of 100 upon reaching the goal, the player wants to reach the goal as fast as possible.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>4 of the maps used in the experiments.</p>
</caption>
<graphic xlink:href="frobt-12-1512099-g001.tif">
<alt-text content-type="machine-generated">Four-panel grid illustrating an agent's position in a maze with obstacles labeled &#x201c;Wumpus.&#x201d; Each of the four illustrations demostrates different location relationships between the Wumpus: relevant, not relevant, blocking, and blocked.</alt-text>
</graphic>
</fig>
<p>Three static monsters are randomly distributed in the 23 tiles other than the start and the goal. The player&#x2019;s objective is to avoid the monsters, because a penalty of 100 is incurred if the player steps into each monster. Stepping into a monster only results in a penalty and does not end the game. Importantly, the locations of the monsters are unknown to and cannot be perceived by the player. This lack of information is designed to motivate communication.</p>
<p>While the player is not aware of the locations of the monsters or able to take any actions to observe them, a helper who knows the locations of the monsters tries to assist the player. To highlight the sparsity of human communication, the helper can only point to one monster on the map when the player is at the starting point.</p>
<p>In this immediate and perilous situation that requires on-the-spot communication, the information communicated is far less than what is needed. We build a model for relevance and want to test whether the most relevant signal predicted by the model is consistent with the choice of the helper.</p>
<p>With the task as a running example, we define our relevance model. We start from the basics of modeling single-agent decision-making with a POMDP, then build a model of two-agent communication on top of them.</p>
<p>We model the player as a POMDP agent. In a POMDP, the world is defined as a state <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In our example, it is a tuple containing the location of the player himself and the three locations of the monsters. The agent also has a set of actions, which are moving left, right, up or down in our game. After taking an action <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the agent transition to the next state <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> with a probability modeled by a transition model <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> while receiving a reward <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In this game, the player moves to an adjacent tile in the chosen direction unless the move would result in hitting a wall, in which case the player remains stationary. The reward structure is based on monster positions: each move costs 1 point; reaching the goal yields 100 points and ends the game; and colliding with a monster results in a 100-point penalty.</p>
<p>Utility is defined as a function that represents the desirability of a state. In our example, the desirability of a state is not only determined by its utility right now, but also the future consequences of actions the agent would take from this state. For an action <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in a state <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the utility can be written as <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. A completely rational agent will take the action with the highest utility. Thus, the value of the state will be the maximum utility of the state under all possible actions <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x5c;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In an optimal MDP policy, the agents take the action with the highest utility <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. However, to model the variability in human decision-making, we use a Boltzmann (or softmax) policy, which is commonly applied in studies of human decision-making (<xref ref-type="bibr" rid="B3">Baker et al., 2011</xref>; <xref ref-type="bibr" rid="B15">Jara-Ettinger et al., 2016</xref>):<disp-formula id="e1">
<mml:math id="m11">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x221d;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a parameter controlling rationality, usually large. If <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is higher, the agent is more strict in taking the action with the highest reward.</p>
<p>When the agent does not know the exact state, like the player in the game, it maintains a belief, a probability distribution over all possible states (<xref ref-type="bibr" rid="B18">Kaelbling et al., 1998</xref>). We can use <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to represent the probability of the current state being <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> according to belief <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In our task, the belief is a discrete probability distribution over all possible tuples of monster locations and self-location. As an example, for a player at the starting point, the 3 monsters will be located in the other 23 tiles, resulting in <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mspace width="0.1em"/>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mspace width="-0.14em"/>
<mml:mn>23</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> possible states. Assuming that the player has no knowledge of the monster locations, we can apply a flat prior: the probability will be <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mspace width="0.1em"/>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mspace width="-0.14em"/>
<mml:mn>23</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> for each state. If the player knows the locations of the monsters <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and his location <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, then the state can be represented by a tuple <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. His belief is <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the Dirac delta function. After the agent taking each action, it receives an observation <inline-formula id="inf25">
<mml:math id="m26">
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> from the environment, which in a Bayesian way updates its belief with a likelihood function <inline-formula id="inf26">
<mml:math id="m27">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The player in our game does not receive any observation from the environment. The only change of its belief results from the helper&#x2019;s communication of which tile seats a monster.</p>
<p>Unknowing of the world state, an agent makes decisions based on its belief. We can extend the definitions of <inline-formula id="inf27">
<mml:math id="m28">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf28">
<mml:math id="m29">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to beliefs. In POMDP, the value of a belief can be defined as <inline-formula id="inf29">
<mml:math id="m30">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, while the utility of an action at a certain belief is assessed by considering what observations may result from the actions, and what could be the belief at the next time step. To solve the POMDP is to find the best strategy in a POMDP problem. The exact solution for a POMDP is hard to achieve, however it is possible to compute policies using approximate POMDP solvers such as the ones proposed by <xref ref-type="bibr" rid="B36">Pineau et al. (2003)</xref>, <xref ref-type="bibr" rid="B44">Spaan and Vlassis (2005)</xref>, <xref ref-type="bibr" rid="B20">Kurniawati et al. (2009)</xref>, <xref ref-type="bibr" rid="B41">Silver and Veness (2010)</xref>, <xref ref-type="bibr" rid="B43">Smith and Simmons (2012)</xref>. Since the POMDP agent does not receive any observation in our specific game, we use a special case of POMDP, the QMDP to solve for the optimal policy (<xref ref-type="bibr" rid="B27">Littman et al., 1995</xref>). In QMDP, the belief-action utility function is the state-action utility weighted the agent&#x2019;s belief<disp-formula id="e2">
<mml:math id="m31">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>Although <xref ref-type="disp-formula" rid="e2">Equation 2</xref> and the QMDP approximation only hold for special POMDPs with no observations, our following formulation can be applied to generic POMDP problems with observations incorporated, with <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> approximated by a POMDP solver.</p>
<p>For a given policy <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, the value of a belief under the policy is <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The value of a belief is the maximum utility of the belief under all possible actions <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In our task, if the player at position (1, 3) is certain that a monster is at (2, 3), then his belief is <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mspace width="0.1em"/>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mspace width="-0.14em"/>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> for each state with (2, 3) as a monster and (1, 3) as self-location. With a goal at (2, 4), he will calculate the utility of the action of going to (2, 3) then (2, 4) to be <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, while the utility of the action of going to (1, 4) then (2, 4) to be <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mspace width="0.1em"/>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mspace width="-0.14em"/>
<mml:mn>21</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mfenced>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mspace width="0.1em"/>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mspace width="-0.14em"/>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mfenced>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>100</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. A rational player will be more likely to choose the second path with an adaptation of <xref ref-type="disp-formula" rid="e1">Equation 1</xref>,<disp-formula id="e3">
<mml:math id="m39">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x221d;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>We broaden our decision-making approach to encompass cooperative scenarios involving two agents, where communication plays a crucial role. Initially, we establish a belief system that takes into account not only the physical state of the world but also the beliefs held by the other agent. In our task, the player lacks knowledge of both the physical state and the helper&#x2019;s belief. Therefore, the player&#x2019;s belief <inline-formula id="inf37">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be represented through a joint probability distribution of <inline-formula id="inf38">
<mml:math id="m41">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the helper&#x2019;s belief <inline-formula id="inf39">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. We call this an interactive state <inline-formula id="inf40">
<mml:math id="m43">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as in interactive-POMDP (<xref ref-type="bibr" rid="B10">Gmytrasiewicz and Doshi, 2005</xref>). <inline-formula id="inf41">
<mml:math id="m44">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In our task, the helper knows the world <inline-formula id="inf42">
<mml:math id="m45">
<mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and the player&#x2019;s the belief <inline-formula id="inf43">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, so her belief is <inline-formula id="inf44">
<mml:math id="m47">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The player&#x2019;s belief <inline-formula id="inf45">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a flat prior on both the physical state and the helper&#x2019;s belief. However, since he knows that the helper has full knowledge of the physical state, the helper&#x2019;s belief and the physical state are not independent, but should have the constraint that the helper&#x2019;s belief reflects the true state <inline-formula id="inf46">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Therefore, his belief is <inline-formula id="inf47">
<mml:math id="m50">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mspace width="0.1em"/>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mspace width="-0.14em"/>
<mml:mn>23</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mfenced>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. His belief about the state is the marginal distribution <inline-formula id="inf48">
<mml:math id="m51">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>To efficiently help the player in the game, the helper needs to predict and evaluate the player&#x2019;s future actions. The helper estimates the player&#x2019;s belief, assumes his rationality, and predicts his actions. For example, in <xref ref-type="fig" rid="F1">Figure 1A</xref>, the helper is aware that the player does not know the monster is at (0, 4). She can predict his action to be moving up to the top then right, and the consequence to be hitting the monster. This prediction is a classic ToM application. The key is to use the player&#x2019;s belief, not use the helper&#x2019;s belief. Adapting from <xref ref-type="disp-formula" rid="e3">Equation 3</xref>, the helper can predict the player&#x2019;s actions<disp-formula id="e4">
<mml:math id="m52">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x221d;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>After predicting the player&#x2019;s actions and consequences, the helper is free to use her own belief to evaluate these consequences. By adapting <xref ref-type="disp-formula" rid="e2">Equation 2</xref>, the helper can evaluate the player&#x2019;s action <inline-formula id="inf49">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with her own belief <inline-formula id="inf50">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>,<disp-formula id="e5">
<mml:math id="m55">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>If she predicts that the player is stepping into a monster, the helper will think the player is in a bad shape. She can then simulate what would happen if the player knew about the monster. If the player can avoid because of the access to this knowledge, then the helper&#x2019;s knowledge is relevant as it improves the player&#x2019;s wellbeing.</p>
<p>Combining the action prediction (<xref ref-type="disp-formula" rid="e4">Equation 4</xref>) and evaluation (<xref ref-type="disp-formula" rid="e5">Equation 5</xref>) steps, if the helper can predict the player&#x2019;s actions <inline-formula id="inf51">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from his belief <inline-formula id="inf52">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, then she can evaluate <inline-formula id="inf53">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with her own belief <inline-formula id="inf54">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="e6">
<mml:math id="m60">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>
<xref ref-type="disp-formula" rid="e6">Equation 6</xref> is important because it shows that the helper can evaluate the players belief based on her own belief of the world. In our task, the helper always knows better of the world than the player, so she can always do better if she were in the player&#x2019;s position. The difference between the helper&#x2019;s evaluation of the player&#x2019;s belief and her own belief is the motivation for the helper to communicate with the player. Only when there is a difference, the helper sees herself as capable of saying anything relevant. Therefore, we can define this difference in utility as the relevance of the helper&#x2019;s belief to the player&#x2019;s belief.<disp-formula id="e7">
<mml:math id="m61">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>The relevance of the helper&#x2019;s belief (<xref ref-type="disp-formula" rid="e7">Equation 7</xref>) is the maximum amount they can help. In real-life, while a helper may want to pass along excessive information to a person, she can only change the person&#x2019;s belief with an act of sparse communication.</p>
<p>Defining the relevance of a belief is critical, but not enough. In real world, we cannot directly copy our belief to the other. The change of belief must go through the interpretation of an utterance. That&#x2019;s why we need to define the relevance of an utterance. For an utterance <inline-formula id="inf55">
<mml:math id="m62">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we represent the belief of its receiver to be updated to <inline-formula id="inf56">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> from the original belief <inline-formula id="inf57">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. In our task, three utterances are available: each of the three monsters. The interpretation is clear: a monster is there. The improvement in utility caused by the utterance is<disp-formula id="e8">
<mml:math id="m65">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>In <xref ref-type="fig" rid="F1">Figure 1D</xref>, if the helper points to the green monster, it is relevant because it helps improve the player&#x2019;s utility. In contrast, pointing to the orange monster is not relevant because it does not improve the player&#x2019;s utility at all. With relevance of a communicative signal calculated with <xref ref-type="disp-formula" rid="e8">Equation 8</xref>, we predict that the rational helper will most likely point to the most relevant monster with the probability<disp-formula id="e9">
<mml:math id="m66">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x221d;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
</sec>
<sec id="s3">
<title>3 Experiments</title>
<sec id="s3-1">
<title>3.1 Measure human navigation prior through inverse reinforcement learning</title>
<p>In order to evaluate the effectiveness of our relevance model, we aim to use it to predict the relevance-based choices of monsters to point, and compare our predictions with the choices of human participants. First, we want to make the decision-making process similar to human participants, so that we can evaluate the beliefs and utterances accurately.</p>
<p>For reinforcement learning agents like a rational agent employing an MDP solver, actions are selected with the same probability if they provide the same future expected values <inline-formula id="inf58">
<mml:math id="m67">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Reflected in our maps, it means if there are two moves giving you the same <inline-formula id="inf59">
<mml:math id="m68">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, these agents will choose them with the same probability regardless of their directions. However, by observing human walking and playing video games, we noted that humans do not always choose between these trajectories evenly. Individuals tend to continue moving in the same direction as their initial movement (<xref ref-type="bibr" rid="B2">Arechavaleta et al., 2008</xref>). Also, they often opt for actions that align geometrically with the goal, choosing movements that bring them closer to the destination.</p>
<p>To capture human preference of moving directions, we utilize the concept of feature expectation within inverse reinforcement learning. This involves formulating the reward as a linear combination of various features associated with the state (<xref ref-type="bibr" rid="B1">Abbeel and Ng, 2004</xref>; <xref ref-type="bibr" rid="B4">Bobu et al., 2020</xref>). Apart from the initial reward function <inline-formula id="inf60">
<mml:math id="m69">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we introduce two additional features. The first feature calculates the angle between the intended action <inline-formula id="inf61">
<mml:math id="m70">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the previous movement direction <inline-formula id="inf62">
<mml:math id="m71">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf63">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The second feature computes the angle between the intended action <inline-formula id="inf64">
<mml:math id="m73">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and a vector <inline-formula id="inf65">
<mml:math id="m74">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> extending from the current position to the goal, <inline-formula id="inf66">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. We introduce two additional features as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>A frame in display of Experiment 1: after participant selects monster at position (2, 3) and completes helpfulness self-rating, the player figure moves to the goal.</p>
</caption>
<graphic xlink:href="frobt-12-1512099-g002.tif">
<alt-text content-type="machine-generated">Two grid diagrams showing an agent located in the bottom-left corner. In both images, the target is in the top-right corner, highlighted in green. In the left grid, the agent has a red arrow labeled &#x22;a1&#x22; pointing up, a blue arrow labeled &#x22;a2&#x22; pointing right, and a black arrow labeled &#x22;v&#x22; pointing up. The right grid shows similar actions, with an additional dashed arrow labeled &#x22;w&#x22; pointing diagonally toward the target.</alt-text>
</graphic>
</fig>
<p>We undertook a pilot experiment to determine the appropriate weights for these two features in influencing human decision-making during navigation. We recruited five participants to navigate the maps in our experiment (shown in <xref ref-type="fig" rid="F1">Figure 1</xref>) but without the monsters (five unique maps). For each trajectory <inline-formula id="inf67">
<mml:math id="m76">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we calculate its reward from the environment <inline-formula id="inf68">
<mml:math id="m77">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the reward from following previous direction <inline-formula id="inf69">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and the reward from moving towards the goal <inline-formula id="inf70">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. We fit a logit model <inline-formula id="inf71">
<mml:math id="m80">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to learn the reward function for the agent (<xref ref-type="bibr" rid="B52">Train, 2009</xref>). With the trajectory collected, we estimate the parameters <inline-formula id="inf72">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2.497</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf73">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5.077</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The reward <inline-formula id="inf74">
<mml:math id="m83">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is only used to predict humans&#x2019; navigation policy. When evaluating the policies, we use the original reward <inline-formula id="inf75">
<mml:math id="m84">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> from the environment.</p>
</sec>
<sec id="s3-2">
<title>3.2 Experiment 1</title>
<p>We assigned participants as the helper in our task, while the player was controlled by a rational AI with QMDP-based approximation (see <xref ref-type="disp-formula" rid="e2">Equation 2</xref>) as policy solution for the POMDP. Our objective is to assess whether a relevance model can effectively predict the human selection of which monster to communicate within a map. We measure the correlation between the model&#x2019;s predictions and the actual choices made by humans.</p>
<sec id="s3-2-1">
<title>3.2.1 Participants</title>
<p>20 undergraduate and graduate students participated in this online study and were compensated with 5 dollar gift cards. Since we newly designed this type of experiment, we do not have a clear estimation of effect size, so we choose 20 as the participant size, which is commonly adopted in lab-based experiments.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Stimuli</title>
<p>We generated a set of 14 distinct maps, each of which can be horizontally mirrored, resulting in a total of 28 maps. Example maps are shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. Among the 14 original maps, a player figure representing the player&#x2019;s starting position is located at position (0, 0). In the mirrored maps, the player&#x2019;s starting position is shifted to (4, 0). To introduce more diverse policies, we incorporated horizontal or vertical barriers in 9 out of the 14 maps. These barriers prevent the player from moving through them, forcing the participants to take a detour.</p>
<p>The 14 maps were manually crafted with the goal of maximizing the variation of relevance in each map, guided by the following principles. First, we ensured that there is always a monster positioned off the shortest paths, like the green monster in <xref ref-type="fig" rid="F1">Figures 1A&#x2013;C</xref>, and the orange monster in <xref ref-type="fig" rid="F1">Figure 1D</xref>. This monster&#x2019;s relevance is intentionally set to be very low: rational agent will not pass through this location. Second, we strategically placed some monsters on a cell that multiple shortest paths pass through, sometimes encompassing all the shortest paths. This monster will have a very high relevance, since the consequences of not knowing its location are undesirable: the player has a high probability of running into this monster.</p>
<p>Third, we introduce the blocking mechanism to show that relevance is not about whether the monster is on a shortest path or not, but on the location-based utility calculation. This manipulation involved an arrangement requiring all the shortest paths passing through the first monster to also pass through the second monster, like the monsters in <xref ref-type="fig" rid="F1">Figure 1D</xref>. On its own, the red monster might have high relevance. However, since all the paths through the red monster (referred to as blocked) must go through the green monster (referred to as blocking), only knowing the red monster does not help the player too much: he will still run into the green monster. In contrast, knowing the green monster helps very much. These two monsters are both on many shortest paths, but the relevance is dramatically different. We assume that to thrive in this type of map, agents need the capacity of forward simulation. They need to think &#x201c;what would happen if I point to that monster instead?&#x201d; to point to the blocking monster instead of the blocked one.</p>
<p>The colors used in the examples are solely for illustration. It is important to note that the monsters are all displayed as the same color in the actual experiment, as in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Features used to model participants&#x2019; innate reward when selecting trajectory. Left: participants tend to choose <inline-formula id="inf76">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>which aligns with momentum <inline-formula id="inf77">
<mml:math id="m86">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>; Right: participants tend to choose <inline-formula id="inf78">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>which directs closer to the goal.</p>
</caption>
<graphic xlink:href="frobt-12-1512099-g003.tif">
<alt-text content-type="machine-generated">Grid-based game view showing two perspectives. On the left, a player's view with a character icon on a blue square, adjacent to a yellow creature in an orange-bordered square. The grid includes a green square. On the right, a partner's view with similar elements, but without the yellow creature visible.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Design and procedure</title>
<p>Participants joined the experiment by accessing a link on their personal computers. The use of mobile phones and tablets was prohibited for this purpose.</p>
<p>Upon joining the experiment, participants were provided with a tutorial that includes an informed consent document. Then, they were given a comprehension quiz to ensure they understood the instructions properly. Prior to the formal trials, participants went through six practice trials. The practice phase included working with three maps and their respective mirror images all presented in a random order.</p>
<p>After completing the practice trials, participants proceeded to the main experiment which comprised of 28 trials. During each trial, the participants were presented with two maps side by side (<xref ref-type="fig" rid="F2">Figure 2</xref>). A large map was displayed on the left side, showing the perspective of the helper (themselves) and revealing the locations of the monsters. The maps are displayed with a size of <inline-formula id="inf79">
<mml:math id="m88">
<mml:mrow>
<mml:mn>8.5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>8.5</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mtext>cm</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. In contrast, the small map on the right provided the player&#x2019;s view for the participant&#x2019;s reference. It did not show the locations of the monsters. The small maps are displayed with a size of <inline-formula id="inf80">
<mml:math id="m89">
<mml:mrow>
<mml:mn>4.5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>4.5</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mtext>cm</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The trials were presented to each participant in a random order, ensuring that the original and mirrored versions of the same map were not shown in two consecutive trials. During the trials, participants need to assist their partner. They can choose to highlight a monster on the large map by double-clicking on the selected monster. The participants&#x2019; choice was then recorded.</p>
<p>After the participants confirmed their choice of highlighted monster, a nine-point Likert scale appeared on the screen. This scale prompts participants to rate the perceived helpfulness of their signal, ranging from 1 (not helpful) to 9 (very helpful). The rating reflects the participant&#x2019;s perception of the helpfulness of the signal, the signal&#x2019;s expected effect on the player, rather than the specific effect in this trial. Upon receiving the highlighted location from the participant, the AI player updates its belief, treating the marked location as a monster. It then updates its policy based on this new information and proceeds to navigate the map accordingly. The navigation route was visually presented to the participants as an animation. Once the player successfully reached the goal, a feedback box appeared on the screen. This feedback included details about the partner&#x2019;s performance, such as the number of steps the player took, the number of monsters they encountered, and the score they received. After reviewing the feedback, participants proceeded to the next trial.</p>
<p>The participants took an exit survey after completing all the trials. The exit survey is an open question for the participant&#x2019;s strategy and the opinion on the experiment.</p>
</sec>
<sec id="s3-2-4">
<title>3.2.4 Results</title>
<p>
<italic>Choice of pointing</italic>. For each map across participants, we compute the probability of each monster being selected. This probability can be denoted as <inline-formula id="inf81">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Initially, we put all the monsters from all maps together, treating all monsters independently. We calculate the relevance of each monster <inline-formula id="inf82">
<mml:math id="m91">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in each map by using <xref ref-type="disp-formula" rid="e8">Equation 8</xref> and use it to predict the probability of choosing the monster by using <xref ref-type="disp-formula" rid="e9">Equation 9</xref>. <inline-formula id="inf83">
<mml:math id="m92">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">relevance</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x221d;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. With a fitted <inline-formula id="inf84">
<mml:math id="m93">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.023</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the participants&#x2019; choice of pointing can be linearly predicted by a softmax of relevance, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. The correlation between <inline-formula id="inf85">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf86">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">relevance</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf87">
<mml:math id="m96">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>40</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.696</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>005</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. When using <inline-formula id="inf88">
<mml:math id="m97">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">relevance</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to linearly predict <inline-formula id="inf89">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with the equation <inline-formula id="inf90">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">relevance</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the regression coefficient is <inline-formula id="inf91">
<mml:math id="m100">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.302</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>40</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>6.138</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>001</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.485</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Our relevance model can predict human participants&#x2019; choices. Top: Predicted probability from relevance vs. human selection probability; Bottom: Relevance vs. logit of human selection probability for the monster with the highest relevance predicted by the relevance model.</p>
</caption>
<graphic xlink:href="frobt-12-1512099-g004.tif">
<alt-text content-type="machine-generated">Two scatter plots with regression lines. The top plot shows a positive correlation (r &#x3d; 0.696) between selection probability predicted by a relevance model and human participants' selection probability. The bottom plot depicts a positive correlation (r &#x3d; 0.830) between the relevance of the monster and the logit of selection probability by participants. Both plots indicate strong positive linear relationships.</alt-text>
</graphic>
</fig>
<p>Since the selection of all the monsters within the same map are not independent, but mutually exclusive, we then specifically look at the most relevant target within each map. We assume that the higher the relevance, the more likely it is to be picked by the participants. A linear regression analysis shows that the probability of choosing the most relevant monster can be predicted by the exponential of its relevance (<xref ref-type="fig" rid="F4">Figure 4</xref>). In most maps, the monster with the highest relevance is frequently chosen. However, in three maps, no monster has a very high relevance (less than 10), including the monster with the highest relevance. In those maps, we expect the probability of the monster with the highest relevance being chosen to be lower than in other maps, since the most relevant monster is only slightly better than the other two monsters.</p>
<p>We calculate the logit of <inline-formula id="inf92">
<mml:math id="m101">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for the most relevant monster on each map, <inline-formula id="inf93">
<mml:math id="m102">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, with a negligibly small hyperparameter arbitrarily chosen as <inline-formula id="inf94">
<mml:math id="m103">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The correlation between relevance and <inline-formula id="inf95">
<mml:math id="m104">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf96">
<mml:math id="m105">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>12</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.830</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. If we use relevance to linearly predict the logit of <inline-formula id="inf97">
<mml:math id="m106">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with the equation <inline-formula id="inf98">
<mml:math id="m107">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">human</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the regression coefficient is <inline-formula id="inf99">
<mml:math id="m108">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.033</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>12</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5.158</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>001</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.689</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>To further analyze the act of pointing to monsters, we study the choice of monsters that fall into the four categories introduced in the stimuli section. Participants chose the monsters off all shortest paths in only <inline-formula id="inf100">
<mml:math id="m109">
<mml:mrow>
<mml:mn>3.93</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the trials. It is very close to 0 as predicted by the relevance model. In the 7 out of 14 maps where the relevance of the most relevant monster is high <inline-formula id="inf101">
<mml:math id="m110">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mspace width="-0.17em"/>
<mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mspace width="-0.17em"/>
<mml:mn>40</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, participants chose the most relevant monster in <inline-formula id="inf102">
<mml:math id="m111">
<mml:mrow>
<mml:mn>80.71</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the trials. For the maps where blocking occurs, participants chose the most relevant monster in <inline-formula id="inf103">
<mml:math id="m112">
<mml:mrow>
<mml:mn>80.00</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the trials. In <inline-formula id="inf104">
<mml:math id="m113">
<mml:mrow>
<mml:mn>16.67</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the trials, they chose the second most relevant monster, the blocked one.</p>
<p>
<italic>Helpfulness self-rating</italic>. We also examine the participants&#x2019; self-rating of their pointing&#x2019;s helpfulness. When analyzing helpfulness self-rating data, we excluded data from one participant who reported not to correctly understand the rating. It is shown that when the participants pointed to the most relevant monster, they rated the pointing as more useful. We use whether the participant chose the most relevant monster <inline-formula id="inf105">
<mml:math id="m114">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>6.87</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.94</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> or not <inline-formula id="inf106">
<mml:math id="m115">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5.97</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2.01</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to predict their self-rating of helpfulness and fit a fixed-effect linear model <inline-formula id="inf107">
<mml:math id="m116">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mspace width="0.3333em"/>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf108">
<mml:math id="m117">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.640</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>001</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.360</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<italic>Self-report strategies</italic>. We check the participants&#x2019; strategies in the game, which were self-reported after the experiment. As shown in <xref ref-type="table" rid="T1">Table 1</xref>, 5 out of 11 participants who reported their strategy mentioned that they took the player&#x2019;s perspective of making decisions, showing the ToM involvement in the signaling process. 9 out of the 11 participants mentioned action prediction and evaluation with phrases such as &#x201c;shortest path&#x201d;, indicating their use of paternalistic evaluation of beliefs. Notably, one participant reported forward simulation. &#x201c;If I feel two points are both important, I will point to one but give myself a low rating. Because I feel pointing to one is not enough, it may also hit the other one.&#x201d;</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Strategies reported by participants in Experiment 1.</p>
</caption>
<table>
<tbody valign="top">
<tr>
<td align="left">Try to interpret my partner&#x2019;s strategy, and find out the monster which would be most helpful to point out in respond to the partner&#x2019;s strategy</td>
</tr>
<tr>
<td align="left">1. find all possible shortest path from partner&#x2019;s point of view; 2. signal the monster that blocks the most number of paths</td>
</tr>
<tr>
<td align="left">If the monster is right beside the starting point or the goal, I point to that monster. If not, I plan a shortest path myself and see which monster impacts the path the most</td>
</tr>
<tr>
<td align="left">If two monsters are next to each other, I&#x2019;ll choose one of them. If a monster is on the shortest path that the partner is likely to pass, I will point to it and make the partner circumvent to a longer and safer path</td>
</tr>
<tr>
<td align="left">Try to think in my partner&#x2019;s shoes and determine which monster is the most dangerous if I were the partner</td>
</tr>
<tr>
<td align="left">Plan a path in my mind as a player and if that path passes through a monster, I click on that monster</td>
</tr>
<tr>
<td align="left">Choose three shortest paths that I prefer and see which monster impacts them the most</td>
</tr>
<tr>
<td align="left">I tried to use the pointing signal to warn the partner of a direction that could potentially harm its utility the most. It&#x2019;s a pity that I could not point to an empty space since sometimes pointing to an empty space does a better job on that</td>
</tr>
<tr>
<td align="left">If I feel two points are both important, I will point to one but give myself a low rating. Because I feel pointing to one is not enough, it may also hit the other one</td>
</tr>
<tr>
<td align="left">I choose one point. If I say my choice is helpful, that means people can avoid the shortest paths through it. If it&#x2019;s not helpful, then there is another point on the shortest path</td>
</tr>
<tr>
<td align="left">I imagine I&#x2019;m the partner. I will choose one route to the goal that I like best. But this route almost always hits monsters. Back to the perspective of myself, I know the first monster on this route and I will point it out</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2-5">
<title>3.2.5 Discussion</title>
<p>The linear relationship between relevance and the logit of the participants&#x2019; choice probability indicates that humans utilize utility-based relevance when deciding which information to communicate. Evidenced by the ratings, humans possess an awareness of the degree of helpfulness associated with each piece of information and thus tend to offer the most beneficial information. The participants&#x2019; self-reported strategies clearly show that humans engage in action prediction and evaluation when assessing relevance. Human participants also incorporate counterfactual reasoning, although they do not explicit express it in their self-reflections.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Experiment 2</title>
<p>We flip the role of the human participant and AI in Experiment 1. We assigned participants as the player in our task, while the helper is a rational AI who communicates either with our relevance model or a heuristics model. The heuristic model points to the monster closest to the starting point as measured by Manhattan distance. For example, in <xref ref-type="fig" rid="F1">Figure 1C</xref>, the heuristic model will point to (0, 2). In 10 out of the 28 maps, the heuristic model points to the same monster as the relevance model.</p>
<p>Our goal is to test a) whether a relevance model can contribute more to task performance than a heuristic model and b) whether the help from a relevance model is better received by human participants.</p>
<sec id="s3-3-1">
<title>3.3.1 Participants</title>
<p>21 undergraduate students participated in this study and were compensated with course credits. One participant was excluded from the analysis, because all their helpfulness ratings were marked as 1.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Stimuli</title>
<p>The maps were the same as in Experiment 1.</p>
</sec>
<sec id="s3-3-3">
<title>3.3.3 Procedure</title>
<p>Participants entered the experiment room and opened the link to the experiment on a computer. Then, participants went through a practice phase similar to Experiment 1. After the practice phase, the participants started the formal 28 trials, arranged in the same manner as Experiment 1. In each formal trial, participants saw the map without monsters, displayed <inline-formula id="inf109">
<mml:math id="m118">
<mml:mrow>
<mml:mn>8.5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>8.5</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mtext>cm</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> in the center of the screen. After 3&#x2013;5 s, a highlight on a cell showed the monster marked by the helper. The participant then controlled the keyboard to navigate the map to the goal. The reward gained by the participants was recorded.</p>
<p>After the participant reached the goal, all the monsters were revealed on the map. A feedback box appeared, showing the participant how many steps they took, how many monsters they ran into, and the score they received. The feedback box also contained a nine-point Likert scale asking the participants to rate the helpfulness of the signal (from 1-not helpful to 9-very helpful). The rating was also recorded for analysis.</p>
</sec>
<sec id="s3-3-4">
<title>3.3.4 Results</title>
<p>Reward. We compare the reward from relevance model and the heuristic model. A paired t-test shows that the participants achieve higher reward when receiving help from the relevance model <inline-formula id="inf110">
<mml:math id="m119">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>54.636</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>17.949</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> than when receiving help from the heuristic model <inline-formula id="inf111">
<mml:math id="m120">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>42.879</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>12.352</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf112">
<mml:math id="m121">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>19</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2.761</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>05</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Specifically, we compare the reward received in the 18 maps where the relevant helper and the heuristic helper point to different monsters. In these 18 maps, a paired t-test shows that the participants achieve higher reward when receiving help from the relevance model <inline-formula id="inf113">
<mml:math id="m122">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>59.111</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>19.300</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> than when receiving help from the heuristic model <inline-formula id="inf114">
<mml:math id="m123">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>40.178</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>16.006</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf115">
<mml:math id="m124">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>19</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3.903</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<italic>Helpfulness rating</italic>. We compare the ratings received from the participants to the relevance model and the heuristic model. A paired t-test shows that the participants give higher ratings to the help from relevance model <inline-formula id="inf116">
<mml:math id="m125">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5.593</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.53</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> than to the help from the heuristic model <inline-formula id="inf117">
<mml:math id="m126">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5.129</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.47</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf118">
<mml:math id="m127">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>19</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3.652</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>005</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Furthermore, we compare the rating received by participants in the 18 maps where the relevant helper and the heuristic helper point to different monsters. In these 18 maps, a paired t-test shows that the participants give higher ratings to the help from the relevance model <inline-formula id="inf119">
<mml:math id="m128">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5.706</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.697</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> than to the help from the heuristic model <inline-formula id="inf120">
<mml:math id="m129">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4.983</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.404</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf121">
<mml:math id="m130">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>19</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3.120</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>01</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Experiment 3</title>
<p>Studies have argued that large language models (LLMs) show ToM as well as planning ability, or even navigation (<xref ref-type="bibr" rid="B40">Sap et al., 2022</xref>; <xref ref-type="bibr" rid="B12">Guo et al., 2023</xref>; <xref ref-type="bibr" rid="B48">Sumers et al., 2023b</xref>; <xref ref-type="bibr" rid="B53">Wang et al., 2024</xref>; <xref ref-type="bibr" rid="B26">Lin et al., 2024</xref>; <xref ref-type="bibr" rid="B35">Park et al., 2023</xref>), by turning planning games into a language game. Since we have a task demanding a coordination of mind reasoning, we are interested in how GPT may perform. We turn our task in Experiment 1 to a language game and use the state-of-the-art large language model (LLM) GPT-4 (<xref ref-type="bibr" rid="B33">OpenAI et al., 2023</xref>) as a participant. GPT-4 was provided with a description of the game and was asked to choose one of the three monsters to point to.</p>
<sec id="s3-4-1">
<title>3.4.1 Stimuli</title>
<p>An example of the prompt we provided is: &#x201c;You are playing a two-person game. A player navigates a <inline-formula id="inf122">
<mml:math id="m131">
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> grid map to a goal. In the map, there are three monsters invisible to the player. The player can see the location of the goal. When the player reaches the goal, the game ends. The player can also see the walls on the map that they cannot go through. The walls are between 2 cells in the map. The player going one step costs 1 dollars. Reaching the goal grants 100 dollars and running into each monster costs 100 dollars. We want to get the highest reward. You play as a helper who can inform the player of the location of only one monster. Do you understand the game?&#x201d;</p>
<p>After GPT-4 repeated the correct rules, we entered the second prompt: &#x201c;If the player starts from (0, 0), the goal is (2, 4), no walls are in the map, the monsters are at (1, 2), (2, 0), and (4, 2). Which monster will you tell the player?&#x201d;</p>
<p>We recorded GPT-4&#x2019;s responses and compared them with the human data collected from Experiment 1.</p>
</sec>
<sec id="s3-4-2">
<title>3.4.2 Results</title>
<p>Choice of pointing. We compared GPT-4&#x2019;s choices with the human participants&#x2019; from Experiment 1.</p>
<p>
<inline-formula id="inf123">
<mml:math id="m132">
<mml:mrow>
<mml:mn>42.1</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of human participants&#x2019; choices are consistent with the choices from GPT-4. In comparison, <inline-formula id="inf124">
<mml:math id="m133">
<mml:mrow>
<mml:mn>58.2</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of human participants&#x2019; choices are consistent with the choices from our relevance model. Humans&#x2019; choices are more consistent with our relevance model than GPT-4&#x2019;s prediction, <inline-formula id="inf125">
<mml:math id="m134">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5.379</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mn>001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>We analyzed the pointing to each monster in the four categories in more detail (<xref ref-type="fig" rid="F5">Figure 5</xref>). In 13 of the 14 maps, GPT-4 will not point to the monster off any shortest path.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>4 of the maps used in the experiments. <bold>(A)</bold> Orange monster high in relevance while green monster low in relevance; <bold>(B)</bold> Red monster high in relevance while orange and green monsters low in relevance; <bold>(C)</bold> Orange monster blocked by red monster; <bold>(D)</bold> Red monster blocked by green monster.</p>
</caption>
<graphic xlink:href="frobt-12-1512099-g005.tif">
<alt-text content-type="machine-generated">Bar chart comparing the percentage of selection for different monster types by Humans, Relevance model, and GPT-4. The categories are &#x22;No shortest path through,&#x22; &#x22;High relevance,&#x22; &#x22;Blocking,&#x22; and &#x22;Blocked.&#x22; The Relevance model shows the highest selection for all categories except &#x22;Blocked,&#x22; where GPT-4 is highest. Humans consistently show varying levels across categories, consistent with the predictions of the relevance model.</alt-text>
</graphic>
</fig>
<p>Is GPT-4 looking at the most valuable monster target? In the seven maps where one monster is very relevant (the calculated relevance of a monster is over 40), GPT-4 only chose the same monster as humans do in <inline-formula id="inf127">
<mml:math id="m136">
<mml:mrow>
<mml:mn>30.4</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the trials, far from the consistency between the relevance model and human participants <inline-formula id="inf128">
<mml:math id="m137">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>80.7</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. This shows that GPT is aware of what the shortest paths are, and that monsters not on any shortest paths is not relevant.</p>
<p>In the maps where one monster blocks another, GPT-4 chose the most relevant monster with a probability of <inline-formula id="inf129">
<mml:math id="m138">
<mml:mrow>
<mml:mn>25</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In the other <inline-formula id="inf130">
<mml:math id="m139">
<mml:mrow>
<mml:mn>75</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the trials, it chose the blocked monster. For example, in <xref ref-type="fig" rid="F1">Figure 1D</xref>, where the red monster is blocked by the green one. GPT-4 points to the red monster, which is rarely consistent with human choice. As we speculated when we designed the maps, even though recent studies shows evidence for LLMs using counterfactual reasoning (<xref ref-type="bibr" rid="B57">Zhang et al., 2024</xref>), their counterfactual reasoning cannot not be spontaneously utilized.</p>
<p>Self-report strategies. GPT-4 also provided legit and reasonable strategies that were similar to the human participant reports in Experiment 1. An example is shown in <xref ref-type="table" rid="T2">Table 2</xref> The answers clearly involve action prediction and evaluation. GPT-4 assumes one trajectory for the player and points to a monster on the trajectory. However, it does not consider all of the player&#x2019;s possible policies. We see human participants report this strategy too, but their choices of pointing to monsters are different from GPT-4.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Examples of strategies reported by GPT-4 in Experiment 3.</p>
</caption>
<table>
<tbody valign="top">
<tr>
<td align="left">Given the new configuration, the hunter&#x2019;s most efficient path avoiding the walls would be:<break/>Move left 2 times to (2, 0) Move up 2 times to (2, 2)<break/>Move left 2 times to (0, 2)<break/>Move up 2 times to (0, 4)<break/>Move right 2 times to (2, 4), the goal Following this path, the monster at position (2, 0) is the one directly in the hunter&#x2019;s way. Therefore, you should warn the hunter about the monster at (2, 0)</td>
</tr>
<tr>
<td align="left">Given the walls and monsters&#x2019; positions, the best monster to reveal to the hunter would be the one at position (3, 1)<break/>Here&#x2019;s why: Monster at (3, 1): This monster is directly on the hunter&#x2019;s potential path towards the goal. Given the walls, the hunter might decide to move left and then up, colliding with this monster<break/>Monster at (3, 2): This monster is also on a potential path to the goal, but the wall between (3, 2) and (2, 2) will prevent the hunter from moving in this direction<break/>Monster at (0, 4): This monster is far from the hunter&#x2019;s starting position and the goal. It&#x2019;s unlikely the hunter would move this far to the left, so revealing this monster would not provide a significant benefit<break/>The optimal path, if the monster at (3, 1) is revealed, would be:<break/>The hunter starts from (4, 0), moves up to (4, 1), then moves up to (4, 2), then moves left to (3, 2), moves up to (3, 3), then moves left to (2, 3), and finally moves up to the goal at (2, 4)<break/>This path avoids all monsters and costs $7. So, the monster at (3, 1) is the best choice to reveal to the hunter</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4-3">
<title>3.4.3 Discussion</title>
<p>Overall in the experiment, GPT-4 performs better than chance level, indicating that it owns some capacity needed to solve the game. However, the consistency of GPT-4&#x2019;s choices with human participants is lower than the consistency of relevance model with participants, indicating that they fall short in planning or ToM ability, compared with human level. Surprisingly, we found that the report of GPT is very similar to human self-report. GPT-4 encompasses the concepts of action prediction and action evaluation, showing a certain level of capacity for ToM and planning. However, GPT-4 is not using the action prediction and action evaluation components to make choices, as humans do. It is more likely that GPT-4 just mimics human thinking processes on a surface level.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 General discussion</title>
<p>Through our experiments, we present concrete evidence that impromptu human communication within cooperative contexts aligns with the principle of utility-based relevance. In our first experiment, we observe that the choices made by human participants regarding which monster to point to can be accurately predicted by our relevance model. Notably, the consistency between the participants&#x2019; choices and the model&#x2019;s predictions remains high across various types of monsters, including those with very low relevance, very high relevance, blocked paths, and those blocking paths.</p>
<p>These findings suggest that when individuals decide what information to convey, they understand the relevance associated with each piece of information. Then they rationally choose the relevant ones. RSA proposes that each utterance in communication carries a utility based on how it alters the listener&#x2019;s belief towards the true belief in a reference game (<xref ref-type="bibr" rid="B7">Frank and Goodman, 2012</xref>). In our experiment, we expand the concept of utility to relevance: the utility of a communicative signal within a task can be defined by how it enhances the listener&#x2019;s performance. The findings are consistent with recent studies on relevance (<xref ref-type="bibr" rid="B46">Sumers et al., 2021</xref>; <xref ref-type="bibr" rid="B17">Jiang et al., 2022</xref>; <xref ref-type="bibr" rid="B16">Jiang et al., 2021</xref>). This seemingly simple utility calculation indicates that humans maintain a representation and plan in the physical world, which largely influences their communication (<xref ref-type="bibr" rid="B8">Friston et al., 2021</xref>).</p>
<p>In Experiment 2, we observe that communication generated by AI integrated with a relevance model yields enhanced performance and receives higher ratings in terms of perceived helpfulness. We aspire that our model can serve as a source of inspiration for designing AI systems that offer both impromptu and efficient interactions with humans.</p>
<p>An AI system equipped with a relevance model only requires planning capacity grounded in decision-making theory to effectively provide concise assistance to humans. This approach could potentially alleviate the intensive training needed for AI to communicate in human natural language or develop a new language (<xref ref-type="bibr" rid="B23">Lazaridou and Baroni, 2020</xref>; <xref ref-type="bibr" rid="B13">Havrylov and Titov, 2017</xref>). Additionally, by answering the question of what to communicate (<xref ref-type="bibr" rid="B39">Roth et al., 2006</xref>), an AI agent may circumvent the necessity of sharing its entire observation from the sensor (<xref ref-type="bibr" rid="B38">Reddy et al., 2021</xref>) with humans. Instead, it can pick the most relevant information in the observation to avoid excess information overload in the cases of large observations, such as images or videos. A recent example is the improvement in human-robot joint task performance by communicating goals in (<xref ref-type="bibr" rid="B24">Le Guillou et al., 2023</xref>), which is highly relevant in cooperative tasks (<xref ref-type="bibr" rid="B50">Tang et al., 2022</xref>).</p>
<p>The strategy descriptions from human participants in Experiment 1 and GPT-4 in Experiment 3 both reflect elements of action prediction and action evaluation. It may leave the impression that GPT-4 possesses ToM abilities similar to humans (<xref ref-type="bibr" rid="B40">Sap et al., 2022</xref>). However, such similarity in language-described strategy is not supported by their choice of signals. In other words, while GPT-4 may sound like a human, it does not always behave like one (<xref ref-type="bibr" rid="B28">Mahowald et al., 2023</xref>). Human decisions across all four monster types align well with action prediction and action evaluation as modeled by the relevance framework, but the behavior of GPT-4 only agrees with the ToM model on one type. This discrepancy might stem from gaps between human reports and human behavior: Since GPT-4 is trained from language data, it may fail to learn certain reasoning processes if they are not frequently articulated in human language. Supporting this hypothesis, in Experiment 1, human choices reflect counterfactual reasoning consistent with the prediction of our relevance model, yet this reasoning is rarely made explicit in the participants&#x2019; verbal reports. This suggests the need for more direct evaluations of behavior of LLMs as agents: other than its language output as a chatbot, we should also focus on its actions.</p>
<p>In the thriving trend of reinforcement learning with human feedback (<xref ref-type="bibr" rid="B34">Ouyang et al., 2022</xref>), it is pivotal to appreciate the value of incorporating behavioral data into model training and conducting data analysis guided by theories in cognitive science. We hope to inspire more interdisciplinary research combining ToM and AI models like POMDP.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The datasets analyzed for this study can be found in the Github repository: <ext-link ext-link-type="uri" xlink:href="https://github.com/kaiwenj/relevanceCommunication">https://github.com/kaiwenj/relevanceCommunication</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Office of the Human Research Protection Program, UCLA. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>KJ: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Project administration, Software, Supervision, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. BJ: Conceptualization, Data curation, Investigation, Software, Writing &#x2013; original draft. AS: Investigation, Writing &#x2013; original draft, Writing &#x2013; review and editing. RL: Investigation, Writing &#x2013; original draft, Writing &#x2013; review and editing. TG: Conceptualization, Funding acquisition, Methodology, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. TG is supported by ONR Vision Language Integration grant (PO&#x23; 951147:1).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>A. Y.</given-names>
</name>
</person-group> (<year>2004</year>). &#x201c;<article-title>Apprenticeship learning via inverse reinforcement learning</article-title>,&#x201d; in <source>Proceedings of the twenty-first international conference on machine learning</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Greiner</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Schuurmans</surname>
<given-names>D.</given-names>
</name>
</person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>1</fpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arechavaleta</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Laumond</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Hicheur</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Berthoz</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>An optimality principle governing human walking</article-title>. <source>IEEE Trans. Robotics</source> <volume>24</volume>, <fpage>5</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2008.915449</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Baker</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Saxe</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Bayesian theory of mind: modeling joint belief-desire attribution</article-title>,&#x201d; in <conf-name>Proceedings of the annual meeting of the cognitive science society</conf-name>, <conf-loc>Boston, Massachusetts, USA</conf-loc>, <conf-date>20-23 July 2011</conf-date>, <fpage>33</fpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bobu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Scobee</surname>
<given-names>D. R. R.</given-names>
</name>
<name>
<surname>Fisac</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Sastry</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Dragan</surname>
<given-names>A. D.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Less is more: rethinking probabilistic models of human behavior</article-title>,&#x201d; in <conf-name>Proceedings of the 2020 ACM/IEEE International Conference on Human-Robot Interaction</conf-name>, <conf-loc>Cambridge, United Kingdom</conf-loc>, <conf-date>23-26 March 2020</conf-date> (<publisher-name>HRI</publisher-name>), <fpage>429</fpage>&#x2013;<lpage>437</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Souza</surname>
<given-names>P. E.</given-names>
</name>
<name>
<surname>Chanel</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Mailliez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dehais</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting human operator&#x2019;s decisions based on prospect theory</article-title>. <source>Interact. Comput.</source> <volume>32</volume>, <fpage>221</fpage>&#x2013;<lpage>232</lpage>. <pub-id pub-id-type="doi">10.1093/iwcomp/iwaa016</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Franco</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gagliano</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Toddlers&#x2019; pointing when joint attention is obstructed</article-title>. <source>First Lang.</source> <volume>21</volume>, <fpage>289</fpage>&#x2013;<lpage>321</lpage>. <pub-id pub-id-type="doi">10.1177/014272370102106305</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Frank</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Goodman</surname>
<given-names>N. D.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Predicting pragmatic reasoning in language games</article-title>. <source>Science</source> <volume>336</volume>, <fpage>998</fpage>. <pub-id pub-id-type="doi">10.1126/science.1218633</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friston</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Moran</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Nagai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Taniguchi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gomi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>World model learning and inference</article-title>. <source>Neural Netw.</source> <volume>144</volume>, <fpage>573</fpage>&#x2013;<lpage>590</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2021.09.011</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gmytrasiewicz</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Adhikari</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Optimal sequential planning for communicative actions: a bayesian approach</article-title>,&#x201d; in <source>Proceedings of the 18th international conference on autonomous agents and MultiAgent systems volume, 19</source> (<publisher-loc>Richland, SC</publisher-loc>: <publisher-name>International Foundation for Autonomous Agents and Multiagent Systems</publisher-name>), <fpage>1985</fpage>&#x2013;<lpage>1987</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gmytrasiewicz</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Doshi</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>A framework for sequential planning in multi-agent settings</article-title>. <source>J. Artif. Intell. Res.</source> <volume>24</volume>, <fpage>49</fpage>&#x2013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1613/jair.1579</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Grice</surname>
<given-names>H. P.</given-names>
</name>
</person-group> (<year>1975</year>). &#x201c;<article-title>Logic and conversation</article-title>,&#x201d; in <source>Speech acts</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Cole</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Morgan</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<publisher-name>Brill</publisher-name>), <fpage>41</fpage>&#x2013;<lpage>58</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>B. Y.</given-names>
</name>
<name>
<surname>Iwasawa</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Matsuo</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Suspicion-agent: playing imperfect information games with theory of mind aware gpt-4</source>. <comment>arXiv preprint arXiv:2309.17277</comment>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Havrylov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Titov</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Emergence of language with multi-agent games: learning to communicate with sequences of symbols</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>30</volume>. <pub-id pub-id-type="doi">10.48550/arXiv.1705.11192</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Littman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>MacGlashan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cushman</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Austerweil</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Showing versus doing: teaching by demonstration</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>29</volume>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jara-Ettinger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gweon</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Schulz</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The na&#xef;ve utility calculus: computational principles underlying commonsense psychology</article-title>. <source>Trends Cognitive Sci.</source> <volume>20</volume>, <fpage>589</fpage>&#x2013;<lpage>604</lpage>. <pub-id pub-id-type="doi">10.1016/j.tics.2016.05.011</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Stacy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rossano</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Individual vs. joint perception: a pragmatic model of pointing as smithian helping</article-title>. <source>Proc. Annu. Meet. Cognitive Sci. Soc.</source> <volume>43</volume>, <fpage>1781</fpage>&#x2013;<lpage>1787</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Stacy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dahmani</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rossano</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>What is the point? a theory of mind model of relevance</article-title>. <source>Proc. Annu. Meet. Cognitive Sci. Soc.</source> <volume>44</volume>, <fpage>3669</fpage>&#x2013;<lpage>3675</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Littman</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Cassandra</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Planning and acting in partially observable stochastic domains</article-title>. <source>Artif. Intell.</source> <volume>101</volume>, <fpage>99</fpage>&#x2013;<lpage>134</lpage>. <pub-id pub-id-type="doi">10.1016/s0004-3702(98)00023-x</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Knox</surname>
<given-names>W. B.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Reinforcement learning from simultaneous human and mdp reward</article-title>,&#x201d; in <source>Proceedings of the 11th international conference on autonomous agents and multiagent systems - volume 1</source> (<publisher-loc>Richland, SC</publisher-loc>: <publisher-name>International Foundation for Autonomous Agents and Multiagent Systems</publisher-name>), <fpage>475</fpage>&#x2013;<lpage>482</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kurniawati</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>W. S.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>SARSOP: efficient point-based POMDP planning by Aapproximating optimally reachable belief spaces</article-title>,&#x201d; in <source>Robotics: Science and Systems IV</source> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>), <fpage>65</fpage>&#x2013;<lpage>72</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kurniawati</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yadav</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>An online pomdp solver for uncertainty planning in dynamic environment</article-title>,&#x201d; in <source>Robotics research: the 16th international symposium ISRR</source> (<publisher-name>Springer</publisher-name>), <fpage>611</fpage>&#x2013;<lpage>629</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lam</surname>
<given-names>C.-P.</given-names>
</name>
<name>
<surname>Sastry</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>A pomdp framework for human-in-the-loop system</article-title>,&#x201d; in <conf-name>53rd ieee conference on decision and control</conf-name>, <conf-loc>Los Angeles, CA, USA</conf-loc>, <conf-date>15-17 December 2014</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>6031</fpage>&#x2013;<lpage>6036</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lazaridou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Baroni</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Emergent multi-agent communication in the deep learning era</source>. <comment>arXiv preprint arXiv:2006.02419</comment>.</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Le Guillou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pr&#xe9;vot</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Berberian</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Trusting artificial agents: communication trumps performance</article-title>,&#x201d; in <source>Proceedings of the 2023 international conference on autonomous agents and multiagent systems</source> (<publisher-loc>Richland, SC</publisher-loc>: <publisher-name>International Foundation for Autonomous Agents and Multiagent Systems</publisher-name>).</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Levin</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pieraccini</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Eckert</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>A stochastic model of human-machine interaction for learning dialog strategies</article-title>. <source>IEEE Trans. speech audio Process.</source> <volume>8</volume>, <fpage>11</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1109/89.817450</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>B. Y.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Brahman</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bhagavatula</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Swiftsage: a generative agent with fast and slow thinking for complex interactive tasks</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>.</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Littman</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Cassandra</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
</person-group> (<year>1995</year>). &#x201c;<article-title>Learning policies for partially observable environments: scaling up</article-title>,&#x201d; in <source>Machine learning proceedings 1995</source> (<publisher-name>Elsevier</publisher-name>), <fpage>362</fpage>&#x2013;<lpage>370</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mahowald</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ivanova</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Blank</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Kanwisher</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Fedorenko</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Dissociating language and thought in large language models: a cognitive perspective</source>. <comment>arXiv preprint arXiv:2301.06627</comment>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Misyak</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Noguchi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chater</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Instantaneous conventions: the emergence of flexible communicative signals</article-title>. <source>Psychol. Sci.</source> <volume>27</volume>, <fpage>1550</fpage>&#x2013;<lpage>1561</lpage>. <pub-id pub-id-type="doi">10.1177/0956797616661199</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nikolaidis</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Srinivasa</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Human-robot mutual adaptation in collaborative tasks: models and experiments</article-title>. <source>Int. J. Robotics Res.</source> <volume>36</volume>, <fpage>618</fpage>&#x2013;<lpage>634</lpage>. <pub-id pub-id-type="doi">10.1177/0278364917690593</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nikolaidis</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ramakrishnan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shah</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Efficient model learning from joint-action demonstrations for human-robot collaborative tasks</article-title>,&#x201d; in <source>Proceedings of the tenth annual ACM/IEEE international conference on human-robot interaction</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Adams</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<publisher-loc>New York; NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>189</fpage>&#x2013;<lpage>196</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Oliehoek</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Amato</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <source>A concise introduction to decentralized POMDPs</source>. <publisher-name>Springer</publisher-name>, <fpage>1</fpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<collab>OpenAI</collab>
<person-group person-group-type="author">
<name>
<surname>Achiam</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Adler</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Akkaya</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <source>Gpt-4 technical report</source>. <comment>arxiv 2303.08774</comment>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ouyang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Almeida</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wainwright</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mishkin</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Training language models to follow instructions with human feedback</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>35</volume>, <fpage>27730</fpage>&#x2013;<lpage>27744</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>O&#x2019;Brien</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Morris</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bernstein</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Generative agents: interactive simulacra of human behavior</article-title>,&#x201d; in <conf-name>Proceedings of the 36th annual acm symposium on user interface software and technology</conf-name>, <conf-loc>San Francisco CA USA</conf-loc>, <conf-date>29 October 2023- 1 November 2023</conf-date>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pineau</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Thrun</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2003</year>). &#x201c;<article-title>Point-based value iteration: an anytime algorithm for pomdps</article-title>,&#x201d; in <source>Proceedings of the 18th international joint conference on artificial intelligence</source> (<publisher-loc>San Francisco</publisher-loc>: <publisher-name>Morgan Kaufmann Publishers Inc</publisher-name>), <fpage>1025</fpage>&#x2013;<lpage>1032</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>C. L.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Latent belief space motion planning under cost, dynamics, and intent uncertainty</article-title>,&#x201d; in <conf-name>Robotics: science and systems</conf-name>, <conf-date>July 12-16, 2020</conf-date>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reddy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dragan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Assisted perception: optimizing observations to communicate state</article-title>,&#x201d; in <source>Proceedings of the 2020 Conference on Robot Learning</source> (<publisher-name>PMLR</publisher-name>) <volume>155</volume>, <fpage>748</fpage>&#x2013;<lpage>764</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Roth</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Simmons</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Veloso</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>What to communicate? Execution-time decision in multi-agent pomdps</article-title>,&#x201d; in <source>Distributed autonomous robotic systems 7</source> (<publisher-name>Springer</publisher-name>), <fpage>177</fpage>&#x2013;<lpage>186</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sap</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>LeBras</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Fried</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Neural theory-of-mind? on the limits of social intelligence in large lms</source>. <comment>arXiv preprint arXiv:2210.13312</comment>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Veness</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Monte-carlo planning in large pomdps</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>23</volume>.</citation>
</ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>R. N.</given-names>
</name>
<name>
<surname>Chanel</surname>
<given-names>C. P.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Pomdp-based adaptive interaction through physiological computing</article-title>,&#x201d; in <source>HHAI2022: augmenting human intellect</source> (<publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>IOS Press</publisher-name>), <fpage>32</fpage>&#x2013;<lpage>45</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Smith</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Simmons</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <source>Heuristic search value iteration for pomdps</source>. <comment>arXiv preprint arXiv:1207.4166</comment>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spaan</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Vlassis</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Perseus: randomized point-based value iteration for pomdps</article-title>. <source>J. Artif. Intell. Res.</source> <volume>24</volume>, <fpage>195</fpage>&#x2013;<lpage>220</lpage>. <pub-id pub-id-type="doi">10.1613/jair.1659</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stankiewicz</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Legge</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Mansfield</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Schlicht</surname>
<given-names>E. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Lost in virtual space: studies in human and ideal spatial navigation</article-title>. <source>J. Exp. Psychol. Hum. Percept. Perform.</source> <volume>32</volume>, <fpage>688</fpage>&#x2013;<lpage>704</lpage>. <pub-id pub-id-type="doi">10.1037/0096-1523.32.3.688</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sumers</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Hawkins</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Griffiths</surname>
<given-names>T. L.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Extending rational models of communication from beliefs to actions</source>. <comment>arXiv preprint arXiv:2105.11950</comment>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sumers</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Griffiths</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Hawkins</surname>
<given-names>R. D.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Reconciling truthfulness and relevance as epistemic and decision-theoretic utility</article-title>. <source>Psychol. Rev.</source> <volume>131</volume>, <fpage>194</fpage>&#x2013;<lpage>230</lpage>. <pub-id pub-id-type="doi">10.1037/rev0000437</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sumers</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Narasimhan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Griffiths</surname>
<given-names>T. L.</given-names>
</name>
</person-group> (<year>2023b</year>). <source>Cognitive architectures for language agents</source>. <comment>arXiv preprint arXiv:2309.02427</comment>.</citation>
</ref>
<ref id="B49">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Taha</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mir&#xf3;</surname>
<given-names>J. V.</given-names>
</name>
<name>
<surname>Dissanayake</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>A pomdp framework for modelling human interaction with assistive robots</article-title>,&#x201d; in <conf-name>2011 IEEE International Conference on Robotics and Automation</conf-name>, <conf-loc>Shanghai, China</conf-loc>, <conf-date>09-13 May 2011</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>544</fpage>&#x2013;<lpage>549</lpage>.</citation>
</ref>
<ref id="B50">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Exploring an imagined &#x201c;we&#x201d; in human collective hunting: joint commitment within shared intentionality</article-title>,&#x201d; in <source>Proceedings of the annual meeting of the cognitive science society</source>, <fpage>44</fpage>.</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tomasello</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2010</year>). <source>Origins of human communication</source>. <publisher-name>MIT press</publisher-name>.</citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Train</surname>
<given-names>K. E.</given-names>
</name>
</person-group> (<year>2009</year>). <source>Discrete choice methods with simulation</source>. <publisher-loc>Cambridge, United Kingdom</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>A survey on large language model based autonomous agents</article-title>. <source>Front. Comput. Sci.</source> <volume>18</volume>, <fpage>186345</fpage>. <pub-id pub-id-type="doi">10.1007/s11704-024-40231-1</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pynadath</surname>
<given-names>D. V.</given-names>
</name>
<name>
<surname>Hill</surname>
<given-names>S. G.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>The impact of pomdp-generated explanations on trust and performance in human-robot teams</article-title>,&#x201d; in <source>Proceedings of the 2016 international conference on autonomous agents and multiagent systems</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Jonker</surname>
<given-names>C. M.</given-names>
</name>
</person-group> (<publisher-loc>Richland, SC</publisher-loc> <publisher-name>International Foundation for Autonomous Agents and Multiagent Systems</publisher-name>), <fpage>997</fpage>&#x2013;<lpage>1005</lpage>.</citation>
</ref>
<ref id="B55">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wilson</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sperber</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>Relevance theory</article-title>,&#x201d; in <source>The handbook of pragmatics</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Horn</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ward</surname>
<given-names>G.</given-names>
</name>
</person-group> (<publisher-name>Wiley</publisher-name>), <fpage>606</fpage>&#x2013;<lpage>632</lpage>.</citation>
</ref>
<ref id="B56">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>You</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Colas</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Alami</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Buffet</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Robust robot planning for human-robot collaboration</article-title>,&#x201d; in <conf-name>2023 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <conf-loc>London, United Kingdom</conf-loc>, <conf-date>29 May 2023 - 02 June 2023</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>9793</fpage>&#x2013;<lpage>9799</lpage>.</citation>
</ref>
<ref id="B57">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>What if the tv was off? examining counterfactual reasoning abilities of multi-modal language models</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>, <conf-loc>Paris, France</conf-loc>, <conf-date>02-06 October 2023</conf-date>, <fpage>21853</fpage>&#x2013;<lpage>21862</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>