<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Virtual Real.</journal-id>
<journal-title>Frontiers in Virtual Reality</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Virtual Real.</abbrev-journal-title>
<issn pub-type="epub">2673-4192</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1623764</article-id>
<article-id pub-id-type="doi">10.3389/frvir.2025.1623764</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Virtual Reality</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Classifying interpersonal interaction in virtual reality: sensor-based analysis of human interaction with pre-recorded avatars</article-title>
<alt-title alt-title-type="left-running-head">Arima et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frvir.2025.1623764">10.3389/frvir.2025.1623764</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Arima</surname>
<given-names>Yoshiko</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2798939/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Harada</surname>
<given-names>Yuki</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2400294/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Okada</surname>
<given-names>Mahiro</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>Department of Psychology, Center for Social and Psychological Research of Metaverse, Kyoto University of Advanced Science</institution>, <addr-line>Kyoto</addr-line>, <country>Japan</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2258357/overview">Antonio Sarasa-Cabezuelo</ext-link>, Complutense University of Madrid, Spain</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/505584/overview">Asuka Takai</ext-link>, Advanced Telecommunications Research Institute International (ATR), Japan</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2983457/overview">Mengting Gong</ext-link>, Jinan University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yoshiko Arima, <email>arima.yoshiko@kuas.ac.jp</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>6</volume>
<elocation-id>1623764</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Arima, Harada and Okada.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Arima, Harada and Okada</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>This study investigates human engagement with a non-responsive, pre-recorded avatar in VR environments. Rather than bidirectional collaboration, we focus on unidirectional synchrony from human participants to the avatar and evaluate its detectability using sensor-based machine learning. Using a random forest model, we classified interactions into cooperation, conformity, and competition, achieving an F1 score of 0.89. Feature importance analysis identified hand rotation and head position as key predictors of interaction states. We compared human-human and human interaction with a non-responsive avatar (pre-recorded motion replay) during a joint Simon task by covertly switching collaborators between humans and non-responsive avatars. Using the classification model, a synchrony index was derived from VR motion data to quantify behavioral coordination patterns during joint actions. The classification indexes were associated with higher cooperation in human-human interactions <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.0262</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and greater conformity in human interaction with a non-responsive avatar <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.0034</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The synchrony index was significantly lower in the non-responsive avatar condition <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.001</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, indicating reduced interpersonal synchrony with non-responsive avatars. These findings demonstrate the feasibility of using VR sensor data and machine learning to quantify social interaction dynamics. This study aimed to explore the feasibility of a sensor-based machine learning model for classifying interpersonal interactions in VR, based on preliminary data from small-sample experiments.</p>
</abstract>
<kwd-group>
<kwd>virtual reality</kwd>
<kwd>human activity recognition</kwd>
<kwd>machine learning</kwd>
<kwd>joint simon effect</kwd>
<kwd>interpersonal synchrony</kwd>
</kwd-group>
<contract-sponsor id="cn001">Japan Society for the Promotion of Science <named-content content-type="fundref-id">10.13039/501100001691</named-content>
</contract-sponsor>
<counts>
<page-count count="14"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Virtual Reality and Human Behaviour</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>We applied wearable sensor-based motion analysis and machine learning (decision trees and linear mixed models) to classify interpersonal synchrony patterns under human&#x2013;human and human&#x2013;avatar (non-responsive) conditions. This approach provides quantitative insight into how interaction context modulates synchrony. We propose a synchrony index derived from VR motion and gaze data, which captures behavioral coordination patterns during joint actions. To examine variations in synchrony, we designed an experimental setting in which participants engaged in a joint Simon task while their collaborator was covertly switched between a human and a non-responsive avatar.</p>
</sec>
<sec id="s2">
<title>2 Joint simon experiment</title>
<p>The Simon effect (<xref ref-type="bibr" rid="B15">Simon, 1969</xref>) is a spatial compatibility effect in which a match or mismatch between the spatial location of a stimulus and its response influences behavior. For example, suppose that red or green stimuli appear randomly on the left or right side of a screen as targets, the response is delayed if the button and stimulus positions do not match, whereas if the task is a Go/No-Go task, a delay does not occur. However, when two stimuli are assigned individually to a pair, the Simon effect reappears as if the pair represents a single person, even though each individual&#x2019;s task is identical to that of the Go/No-Go task (<xref ref-type="bibr" rid="B13">Sebanz et al., 2003</xref>). This is known as the joint Simon effect (JSE). Our research team has confirmed that the JSE occurs in VR environments (<xref ref-type="bibr" rid="B6">Harada et al., 2025</xref>). This study utilizes the same VR experimental setting to examine how human-human and human interaction with a non-responsive avatar influence social coordination and synchrony.</p>
<p>Explanatory theories suggest that the JSE reflects either (i) task co-representation or (ii) spatial coding relative to the position of the collaborator (<xref ref-type="bibr" rid="B2">Dolk et al., 2014</xref>). Experiments comparing these two explanatory theories indicate that the latter, i.e., the reference-coding hypothesis (<xref ref-type="bibr" rid="B14">Sellaro et al., 2015</xref>; <xref ref-type="bibr" rid="B12">Sangati et al., 2021</xref>), is supported by more studies. However, as will be discussed below, the co-representation hypothesis is not precluded because the JSE weakens when collaborators are taught that they are unconscious, non-living entities. These hypotheses relate to whether we recognize non-living collaborators merely as reactive entities or as agents capable of shared representations. This pilot study serves as a proof-of-concept for applying sensor-based machine learning models to classify social coordination patterns in immersive virtual environments.</p>
<sec id="s2-1">
<title>2.1 JSE with bot</title>
<p>Previous research on the JSE with non-human collaborators has yielded conflicting results (<xref ref-type="bibr" rid="B19">Stenzel et al., 2016</xref>). reported that the JSE occurs even when the collaborator is a non-living entity. In contrast (<xref ref-type="bibr" rid="B21">Tsai et al., 2008</xref>), analyzed action indices and event-related potentials and found that the JSE emerged only when participants believed their partner to be human.</p>
<p>This discrepancy can be explained by the perception of intentionality (<xref ref-type="bibr" rid="B20">Tsai et al., 2007</xref>; <xref ref-type="bibr" rid="B18">Stenzel et al., 2012</xref>). demonstrated that the JSE is enhanced when the collaborator is perceived as having intentionality, suggesting that recognizing intentionality facilitates action simulation in the motor system. Furthermore (<xref ref-type="bibr" rid="B17">Stenzel and Liepelt, 2014</xref>), found that the perception of agency&#x2014;i.e., seeing another person pressing a button&#x2014;precedes the cognition of that person&#x2019;s intention. Agency can be inferred by observing simple moving figures, even in the absence of explicit social cues (<xref ref-type="bibr" rid="B7">Heider and Simmel, 1944</xref>).</p>
<p>Thus, not all JSE-related co-representation processes rely on higher-order cognition (<xref ref-type="bibr" rid="B9">Miss et al., 2022</xref>; <xref ref-type="bibr" rid="B8">Liepelt et al., 2016</xref>) demonstrated that the JSE was intensified when activity in the anterior cingulate cortex, which is associated with motor intentions, was suppressed. Their findings suggest that when the JSE occurs, the distinction between self and other motor intentions becomes less defined. While the perception of agency is processed automatically through perceptual cues, the intention of others is subsequently inferred.</p>
<p>To further investigate factors influencing automatic JSE processes, this study examines interpersonal synchrony as an indicator of self-other undifferentiated states (<xref ref-type="bibr" rid="B10">Paladino et al., 2010</xref>). Additionally, we explore whether participants recognize the bot as a human in interpersonal synchronization.</p>
</sec>
<sec id="s2-2">
<title>2.2 Interpersonal synchrony</title>
<p>Face-to-face communication evokes a subconscious process of spontaneous synchronization of attention, behavior, and brain waves. A meta-analysis of synchrony studies showed that sensory and interpersonal synchrony resulted in prosocial attitudes and behaviors (<xref ref-type="bibr" rid="B11">Rennung and G&#xf6;ritz, 2016</xref>). As a causal effect in the opposite direction, pro-sociality can promote synchrony. For example (<xref ref-type="bibr" rid="B3">Fronda and Balconi, 2022</xref>), demonstrated that the act of giving affected performance and brain-brain synchrony during cooperative tasks. <xref ref-type="bibr" rid="B16">Smykovskyi et al. (2024)</xref> revealed that negative emotions disrupted intentional synchrony during sensorimotor interactions. Furthermore (<xref ref-type="bibr" rid="B5">Hao et al., 2024</xref>), showed that group identity influenced brain-to-brain synchrony and cooperative decision-making behaviors.</p>
<p>Interpersonal synchrony is assumed to be an automatic process because it occurs within a short reaction time (RT) (<xref ref-type="bibr" rid="B1">Decety et al., 2011</xref>). Synchrony studies have primarily been conducted by measuring the cross-correlation coefficient (CCC) of physiological data. For example (<xref ref-type="bibr" rid="B4">Guastello et al., 2023</xref>), proposed a system in which each member&#x2019;s physiological data was obtained individually, and then cross-correlation was used to distinguish multiple influences on others.</p>
<p>In the present study, we classified interpersonal activities using sensor data related to pairwise units and applied them to human activity recognition (HAR) research. HAR has yielded numerous results through the use of smartphone sensor data and other machine-learning sources to classify activity types, particularly in exercise situations.</p>
</sec>
<sec id="s2-3">
<title>2.3 Social presence</title>
<p>Social presence refers to the perception of being attended to and understood by another entity during an interaction. Recent studies have shown that people can experience social presence even with artificial agents under certain conditions. For example, <xref ref-type="bibr" rid="B22">Chen et al. (2023)</xref> developed and validated a multidimensional scale for assessing robot social presence, expanding traditional dimensions such as physical presence and conscious awareness to include interactional aspects like dialog behavior and emotional understanding. Similarly, <xref ref-type="bibr" rid="B25">Sogemeier et al. (2024)</xref> reported that temporal cues, such as response latency, were more influential than visual realism in eliciting social presence with in-car voice assistants, suggesting that behavioral responsiveness may be more critical than appearance in creating a sense of connectedness. <xref ref-type="bibr" rid="B23">Munnukka et al. (2022)</xref> further demonstrated that perceived anthropomorphism increased social presence in web-based avatar interactions, which in turn fostered trust, although avatar appearance itself had no significant effect on perceived anthropomorphism. Based on these findings, the present study implemented a non-verbal VR bot avatar that replicated recorded human motion but did not engage in speech or dialog. Due to the small sample size, we used only two 7-point Likert-scale items assessing how realistic and human-like the avatar appeared. These items formed a minimal composite index of perceived social presence. At the end of the experiment, participants were also asked to identify in which session they believed they had interacted with a bot. Follow-up interviews were conducted to explore when and how they noticed the bot&#x2014;or whether they failed to detect it at all. This combined qualitative and quantitative data was used to construct a binary variable (&#x201c;Bot-Notice&#x201d;) indicating whether the bot was consciously recognized.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Research question</title>
<p>In this study, we aimed to develop a machine learning-based classification model for interpersonal interactions in VR using sensor data. To validate this model, we examined differences between human-human and human-bot interactions in a joint Simon task.</p>
<p>Research Question 1: To what extent can interpersonal interactions in cooperative tasks be effectively classified using VR sensor data?</p>
<p>As part of our initial exploratory analysis, we conducted a preliminary experiment to classify the activities of pairs in the Simon task based on two basic types of interpersonal interaction in the social sciences: competition and cooperation. We expected that interpersonal synchrony would be more prominent in behaviors classified as cooperative.</p>
<p>The purpose of the preliminary experiment was to discover important features for classifying interpersonal behavior from various sensor data, while determining which phase of the Simon task could be more accurately classified by dividing the task into smaller phases. For this purpose, we used a random forest model, which makes it easy to judge the importance of features, and adopted the most important feature as an indicator of synchrony. Random forest is a machine learning technique widely used in machine learning competitions due to its high prediction accuracy and robustness against overfitting. <xref ref-type="bibr" rid="B24">Sekitani and Murakami (2022)</xref> compared 30 statistical and machine learning models, including their combinations, using symmetric mean absolute percentage error (sMAPE) and mean absolute scaled error (MASE). Their results demonstrated that among individual machine learning methods, Random Forest achieved the highest accuracy. Unlike single decision trees, random forest mitigates overfitting by aggregating multiple trees, enhancing generalization performance. Additionally, it provides an intuitive method for evaluating feature importance, making it a valuable tool for understanding the contribution of each variable in predictive modeling.</p>
<p>Research Question 2: What are the key differences between human-human and human interaction with a non-responsive avatar?</p>
<p>In the main experiment, we established a bot condition in which a bot avatar was introduced as a collaborator and compared the bot condition with a human condition, where participants performed the joint Simon task with a human partner. The bot avatar was created by tracing the sensor data of a human in a preliminary experiment. In the bot condition, synchronization from human to bot is expected, but synchronization from bot to human does not occur. Therefore, it is expected that synchrony in the bot condition will be reduced compared to the human condition. We hypothesized that synchrony would be the key difference between the bot and human conditions, while also exploring other potential differences.</p>
</sec>
<sec sec-type="methods" id="s4">
<title>4 Methods</title>
<sec id="s4-1">
<title>4.1 Preliminary experiment</title>
<sec id="s4-1-1">
<title>4.1.1 Participants</title>
<p>Eight participants (six men and two women; college students aged 19&#x2013;21&#xa0;years) enrolled in the study. The participants were segregated into four groups, with each pair referred to as collaborators.</p>
</sec>
<sec id="s4-1-2">
<title>4.1.2 Participation-agreement procedures</title>
<p>Recruitment was open for 1&#xa0;week from 17 February 2023. Participants were given an explanation of the consent document in the laboratory, and the informed consent procedure was carried out. The participants were handed a paper that outlined the experiment and data-handling procedures, which were explained by the experimenter. All eight participants agreed to participate in the study. The experimental data were obtained using anonymized ID numbers. This ensured that the data were not linked to the participants&#x2019; names.</p>
</sec>
<sec id="s4-1-3">
<title>4.1.3 Devices</title>
<p>The VR systems were established in two separate rooms. Each system comprised a VIVE Pro Eye (HMD), two controllers (VIVE Controller 2018), two base stations (SteamVR Base Station 2.0), and a computer. The VR environment was created using Unity (2021.3.1f1) in a server-client network using &#x201c;Netcode for Game Objects.&#x201d; In this environment, paired participants entered the same virtual space and interacted via physical actions. No audio communication was available, and the VR environment featured two avatars, buttons, a display, and a mirror (<xref ref-type="fig" rid="F1">Figure 1</xref>). All experimental configurations and spatial arrangements shown in <xref ref-type="fig" rid="F1">Figure 1</xref> represent the layout within the virtual reality environment, not the physical laboratory setup. The avatars were able to move based on six-coordinate data (three positions and three rotations) obtained from the HMD and two controllers. These avatars were boxy and lacked personality traits, and their movements were executed using the &#x201c;Final IK (Inverse Kinematics)&#x201d; asset. Red- and green-labeled reaction buttons were placed in front of the avatar in the VR space. The RTs were acquired via collision detection when the avatar touched a button.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Experimental setup for the joint Simon task and conformity/competition tasks. Note: The upper section shows the joint Simon task procedure, while the lower section illustrates the conformity and competition tasks.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g001.tif">
<alt-text content-type="machine-generated">Diagram illustrating joint Simon, conformity, and competition tasks. For each task, two avatars, representing participants, are shown with display and button panels. Phases include: Time Count with a numeral, Fixation Cross with a plus sign, Target with red or green squares, and Blank. Target phase requires button pressing corresponding to square color. The sequence of phases is visually represented with arrows.</alt-text>
</graphic>
</fig>
<p>The task comprised four phases: Phases 1-4 (time count, fixation cross-presentation, target presentation, and blanks, respectively). Phases 1 and 3 were the countdown and motion phases, respectively (See <xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Task phases in the joint Simon task. Note: Phase 1: A 3-s countdown display. Phase 2: Presentation of a black fixation cross, &#x201c;&#x2b;&#x201c;, at the center of the display for 1&#xa0;s. Phase 3: Presentation of targets (red or green) on either the left or right side of the display until a response was obtained. Phase 4: A blank interval of 0.5&#xa0;s before the next countdown began.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g002.tif">
<alt-text content-type="machine-generated">Flowchart depicting a sequence of four phases: Phase 1 is &#x22;Countdown&#x22; lasting three seconds, Phase 2 is &#x22;Fixation cross&#x22; for one second, Phase 3 is &#x22;Target&#x22; until response, and Phase 4 is &#x22;Blank&#x22; for zero point five seconds. Each phase leads into the next with arrows.</alt-text>
</graphic>
</fig>
<p>The participants were instructed to touch a button corresponding to the target color, regardless of its location. Each session comprised 16 or 32 consecutive trials, with the target color and position randomized between the trials.</p>
</sec>
<sec id="s4-1-4">
<title>4.1.4 Procedure</title>
<p>The participants were allowed to select either the client or host experimental booths. The terms &#x201c;host&#x201d; and &#x201c;client&#x201d; were designated because paired data were transmitted as streamed data from the client to the host PC. Following the instructions of the experimenter stationed at each booth, the participants were instructed to wear the HMDs and operate the controllers with both hands. Before each session, the participants were briefed on the colors of the stimuli for which they were responsible. Before commencing the joint task, the participants were instructed to view their collaborators directly and then confirm their avatars in the mirror set in the VR space.</p>
<p>The host stood on the right, whereas the client stood on the left. <xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the experimental setup for the joint Simon task and conformity/competition tasks within the virtual reality environment. Participants interacted with color targets displayed on screen, responding by pressing corresponding buttons in the virtual environment according to task instructions. The host operated the buttons in the VR space using the left hand, whereas the client used the right hand. Thus, the right hand was not used on the host side, and the left hand was not utilized on the client side. The task involved pressing a button labeled with the corresponding color name when the assigned color appeared. The participants were instructed to halt if they felt uncomfortable, lift their HMDs at the end of each session, and take breaks as required.</p>
</sec>
<sec id="s4-1-5">
<title>4.1.5 Sessions</title>
<p>The participants entered the space individually and completed eight practice trials for the Go/No-Go task. During the practice session, the correct answer was indicated when the correct button was touched. An incorrect answer was revealed when another button was touched or when a certain amount of time had elapsed without a touch being detected. If a participant failed in all eight trials, then the practice session was repeated. After the practice session was completed, the following sessions were conducted.</p>
<p>Session 2 involved the procedure shown in the upper section of <xref ref-type="fig" rid="F1">Figure&#xa0;1</xref>. The target colors in Session 3 were swapped to minimize the learning effects. Sessions 4 and 5 involved the procedure shown in the lower section of <xref ref-type="fig" rid="F1">Figure&#xa0;1</xref>. The detailed session structure for the preliminary experiment is presented in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Session structure for the preliminary experiment.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Session</th>
<th align="left">Task</th>
<th align="left">Target assignment</th>
<th align="left">Trials</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">Go/No-Go task</td>
<td align="left">Individual sessions for assigned target</td>
<td align="left">32</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">Joint Simon task</td>
<td align="left">Host: green; Client: red</td>
<td align="left">32</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">Joint Simon task</td>
<td align="left">Host: red; Client: green</td>
<td align="left">16</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">Conformity task</td>
<td align="left">Simultaneous touching</td>
<td align="left">16</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">Competition task</td>
<td align="left">Touch before opponent</td>
<td align="left">16</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Sensor data from Sessions 4 (conformity) and 5 (competition) were used for machine learning and testing, respectively. In the conformity task session, the participants were instructed that &#x201c;whichever target appears, touch the correct button at the same time as your companion.&#x201d; During the competition task session, the participants were instructed that &#x201c;whichever target appears, touch the correct button before your companion.&#x201d;</p>
</sec>
<sec id="s4-1-6">
<title>4.1.6 Data processing</title>
<p>In the preliminary experiment, we investigated the features necessary for distinguishing between interpersonal behaviors in VR environments. To identify subtle differences in subconscious movements, we used the sensor data during Phase 1, i.e., the time at which the participants were staring at the countdown, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The phases and procedures illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref> were all conducted within the virtual reality space.</p>
<p>The sensing data included gaze direction, eye position, pupil size (left and right), head position and rotation, and the position and rotation of the left and right controllers. For each of these, XYZ three-axis data were recorded where applicable. The transmission latency from the client to the host was approximately 0.01&#xa0;s. Signals were sampled at a variable rate (80&#xa0;Hz average), and after missing values were removed, the client and host data were linked at intervals of approximately 0.02&#xa0;s and then used for machine learning. The features used for machine learning were: distance, which was obtained as the root sum of squares of the XYZ (Euler angle) of the position and gyro sensor at each sampling point; the velocity from the time difference; and the acceleration obtained from the time difference in velocity, which was used as the analysis data. After deleting samples with missing values, we used the Python sklearn Random Forest Classifier (n_estimators &#x3d; 250, random_state &#x3d; 42) as the random forest model. A set of decision trees was constructed for a subset of randomly sampled training data, and predictions based on a subset of these features were aggregated to obtain the final prediction. After testing various feature types, we found that distance features yielded relatively high classification accuracy, whereas models using velocity and acceleration performed poorly. Therefore, we decided to use only distance features, except for triaxial gaze data, which were retained as they are considered essential for synchronization. To evaluate classification accuracy, cross-validation was conducted by iteratively designating data from three out of eight participants as the test set, while data from the remaining five were used for training. This process was repeated to ensure that each participant appeared in the test set at least once. The final model was trained on the full dataset after cross-validation.</p>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 Main experiment</title>
<sec id="s4-2-1">
<title>4.2.1 Participants</title>
<p>The participants of this experiment were recruited through a university website, and assigned to each experimental day. Recruitment was open for 2&#xa0;months from 14 May 2023, and the informed consent procedure was the same as the preliminary experiment, but was obtained in advance via a web-based questionnaire in order to avoid coercion in obtaining consent due to face-to-face situations in the laboratory. Owing to the absence of one participant, the final number of participants was 18, which comprised seven men, nine women, and two other genders. The average age of the participants was 19.83 years (standard deviation [SD] &#x3d; 1.04). All participants were assessed for handedness using a self-report questionnaire. Of the 18 participants, 16 were right-handed, 1 was left-handed, and 1 reported being ambidextrous or having no hand preference. After the experiment, the participants were instructed to complete a questionnaire survey and interview, for which they received an honorarium of approximately $10 (1,500 yen) after completion. The device and experimental procedures were identical to those used in the preliminary experiments. Two types of avatars, i.e., a box and a human, were designed for other research purposes (<xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Avatars used in the main experiment. Note: For box-shaped avatars, the leftmost avatar was used regardless of participant gender. For human-shaped avatars, female participants used the middle avatar, while male participants used the left avatar. The avatars shown were generated using VRoid Studio (&#xa9; pixiv Inc.), which permits research use under its license.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g003.tif">
<alt-text content-type="machine-generated">Three digital characters in a progression of detail. On the left, a simple blocky model; in the middle, an anime-style character with a hoodie and dark hair; on the right, a more detailed anime-style character with short dark hair and a white t-shirt. The avatars shown were generated using VRoid Studio (&#xa9; pixiv Inc.), which permits research use under its license.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Sessions</title>
<p>All participants completed the sessions in the same fixed order shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Session structure for the main experiment. Conformity task: participants were instructed to touch the correct target simultaneously. Competition task: participants were instructed to touch the correct target faster than their partner. Joint Simon task: participants were instructed to touch the button only when the target color assigned to them appeared.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Session</th>
<th align="left">Task description</th>
<th align="left">Avatar type</th>
<th align="left">Trials</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">Go/No-Go task (individual)</td>
<td align="left">Box avatar</td>
<td align="left">32</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">Joint Simon task (human-human pair)</td>
<td align="left">Box avatar</td>
<td align="left">32</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">Joint Simon task (human-human pair, target colors swapped)</td>
<td align="left">Human avatar</td>
<td align="left">32</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">Conformity task (human-human pair)</td>
<td align="left">Human avatar</td>
<td align="left">16</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">Competition task (human-human pair)</td>
<td align="left">Human avatar</td>
<td align="left">16</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">Joint Simon task (human vs. Bot pair)</td>
<td align="left">Human avatar</td>
<td align="left">32</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">Joint Simon task (human vs. Bot pair, target colors swapped)</td>
<td align="left">Box avatar</td>
<td align="left">16</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Dependent variable</title>
<p>Correct response rate: Correct responses were counted when participants touched the correct color target in trials where they were required to respond and refrained from touching in trials where they were not. Subsequently, the correct response rate was divided by the number of trials.</p>
<p>JSE: The mean RT delay (RTs for incompatible targets minus RTs for compatible targets) during the joint Simon task (Sessions 2, 3, 6, and 7) minus that of the Go/No-Go task (Session 1) was calculated. The RTs for correct responses with more than two standard deviations from the mean RT were excluded as outliers. Additionally, pairwise data from participants whose RT could not be measured because of equipment failure were excluded.</p>
<p>Bot cognition: After the experiment, a structured assessment combining questionnaires and follow-up face-to-face interviews was conducted to systematically evaluate participants&#x2019; awareness of the bot condition. The assessment protocol was designed to minimize leading questions and retrospective bias. A participant who perceived the human collaborator to be a bot was assumed to be unaware of the discrimination between humans and bots. In the data analysis, binary values of 1 and 0 were used to indicate the awareness and unawareness of bots, respectively. As a proxy for perceived social presence, we also included a two-item measure rated on 7-point Likert scales. The items assessed how realistic and how human-like the avatar appeared. The sum of these two items was used as an index of social presence (Mean &#x3d; 8.78, standard deviation [SD] &#x3d; 2.29).</p>
<p>Sensor data: By performing the procedures of the preliminary experiment, the distance was obtained as the root sum of squares of the XYZ (Euler angle) of the position and gyro sensor at each sampling point; the velocity was the time difference between the two; and the acceleration was the time difference between the two. Because the accuracy of the classification model using velocity and acceleration data was low during the machine learning process, we adopted a model that used only distance features. The resulting features were the HMD position, HMD rotation, and 12 variables of position information for the left- and right-controller positions and rotations (<xref ref-type="fig" rid="F2">Figure 2</xref>). The eye-gaze and pupil size features used in the preliminary experiments were not used in the main experiment because no sensing data corresponded to the bot. Under the bot condition, only trace data from Phase 3 were used; thus, data from Phase 3 were used to formulate the classification model. Samples with missing values on the host or client side were deleted. After removing missing values, the number of observations obtained from Sessions 3, 4, and 5 was 16,759 for training and testing, out of a total of 124,193 observations (approximately 86.9% of the original data were retained). Sensor data from Sessions 2, 6, and 7 were prepared as files on the host and client sides to compare the human and bot conditions.</p>
<p>Pair Activity Probability: Details of machine learning, angular transformations, and statistical models are described in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>.</p>
<p>Synchrony Index: As in the preliminary experiment, the most important feature for classifying paired activities was the rotation of the unused hand (host side right-hand rotation). Therefore, the cross-correlation (CCC) of the sensor data for the host side right-hand rotation and client-side left-hand rotation was calculated for each trial and then used as the interpair synchrony index using MATLAB&#x2019;s XCORR function. After normalizing the sensor data for each trial, the maximum value obtained at lag0 was used as the CCC index.</p>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 Data analysis and modeling procedures</title>
<sec id="s4-3-1">
<title>4.3.1 Machine learning classifier</title>
<p>We used the MATLAB Classification Layer application for the machine learning model to compare the decision trees, random forests, support vector machines (SVMs), and neural nets. The results showed that even a single decision tree provided a correct answer rate exceeding 90%, which is comparable to the performance of other methods. Thus, we adopted a decision tree model to identify the most important features. The Gini diversity index was used as the splitting criterion.</p>
<p>Each of the sensor datasets in the three training sessions was subjected to machine learning, with target variables for classification (10% was used for data verification and cross-validation).</p>
<p>The Receiver Operating Characteristic (ROC) value calculated from the true-positive and false-positive rates exceeded 0.98, which was sufficient for classification accuracy. The ROC curves are presented in the <xref ref-type="sec" rid="s16">Supplementary Appendix</xref>. The final number of branching nodes was 191.</p>
<p>The most important features for classifying joint activities were the right-hand rotation unused by the host and the left-hand rotation unused by the client. The features of the model were similar to those of the preliminary experiments, which classified conformity and competition in the countdown phase, thus suggesting that joint action in the subconscious movement can be classified using the categorization model based on the motion phase.</p>
<p>Using MATLAB&#x2019;s trained predict function, we applied the classification model to Sessions 2, 6, and 7 as test sessions using the 12 feature variables. The classification probability results for each observation were the activity indices of cooperation, conformity, and competition.</p>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Angular transformation and index calculation</title>
<p>These probability values were angularly transformed using <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>arcsin</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mtext>probability</mml:mtext>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>180</mml:mn>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. We adjusted for a probability of 0 by setting <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>arcsin</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mn>0.0833</mml:mn>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>180</mml:mn>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and a probability of 1 by setting <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>arcsin</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.0833</mml:mn>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>180</mml:mn>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. These corrections were performed based on the usual adjustment <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> for angular transformations. Thus, the minimum and maximum possible values were 16.54 and 78.69, respectively. These values were averaged for each trial and used as cooperation, conformity, and competition indices for the pair activity.</p>
<p>The term <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> used in the angular transformation refers to the usual adjustment for proportions, where <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of response options. After applying this correction, the angular transformation of the minimum and maximum possible proportions (0 and 1) results in values of 16.54 and 78.69, respectively. These are dimensionless values resulting from the arcsine transformation of relative proportions; therefore, no physical units are associated with them.</p>
</sec>
<sec id="s4-3-3">
<title>4.3.3 Linear mixed models</title>
<p>Because the human condition comprised data that switched from the client to the host for comparison with the bot condition, we analyzed the condition effects via multilevel analysis. For the linear mixed model, paired groups were specified as random-effect factors after centralization was performed, in which the mean value of each paired group was subtracted from each indicator.</p>
<p>As the Akaike Information Criterion (AIC)s of each indicator&#x2019;s random intercept and random slope models were similar or lower for the random slope model, we report the results for the random slope model here. Considering the few people in the random variable and a p-value that is likely to be high, we report the results of the robust model obtained via the log-likelihood ratio test. Owing to the low overall variance, we report the fixed-factor effects of the mixed model, as well as the results of the test using marginal mean estimation (in contrast to the human condition set to 1 and the bot condition set to 0).</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s5">
<title>5 Results</title>
<sec id="s5-1">
<title>5.1 Preliminary experiment</title>
<p>A random forest model was applied to 21 features selected during the training sessions. The results showed that the confusion matrix between the model predictions and observed data was 88%, and the F1 score was 0.8925 (precision &#x3d; 0.8066; recall &#x3d; 0.9988).</p>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> shows the top features selected by the decision tree classifier in the preliminary experiment. Features with importance <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.10 are listed individually, and all others are summarized in a single row.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Feature importance scores from the decision tree classifier (preliminary experiment). Features with importance <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mo>&#x3c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.10 are summarized in one row.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Importance</th>
<th align="left">Feature name(s)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0.16</td>
<td align="left">Host right-hand rotation</td>
</tr>
<tr>
<td align="left">0.15</td>
<td align="left">Client left-hand rotation</td>
</tr>
<tr>
<td align="left">0.14</td>
<td align="left">Host left-hand position</td>
</tr>
<tr>
<td align="left">0.10</td>
<td align="left">Host right-hand position</td>
</tr>
<tr>
<td align="left">&#x3c;0.10</td>
<td align="left">Host left-hand rotation, Client head position, Host head position</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Host head rotation, Client left-eye pupil size, Client right-eye</td>
</tr>
<tr>
<td align="left"/>
<td align="left">pupil size, Client right-hand position, Client gaze direction</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Host right-eye pupil size, Host gaze (x, y, z), Client left-eye</td>
</tr>
<tr>
<td align="left"/>
<td align="left">pupil size, Client head rotation, Client gaze (x, y, z)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The most important features for classification were the position and rotation of the left and right controllers, followed by the position and rotation of the head-mounted display (HMD), and finally, the gaze and pupillary reflexes. The higher importance of host-side features was probably due to a slight delay in data transmission from the client side. This finding suggests that conformity or competition can be predicted by the twisting motion of the hands of a person who does not touch the button. Therefore, using the normalized variables of the host&#x2019;s right-hand rotation and its counterpart, i.e., the client&#x2019;s left-hand rotation, we calculated the measure of synchrony for each of the four pairs, and the cross-correlation coefficient (CCC) was calculated for each of the four pairs as a measure of synchrony.</p>
<p>Using this conformity or competition classification model, we analyzed how the ratio of conformity or competition status changed during the countdown phase in the joint Simon session. We discovered that the occurrence probability of a category classified as competition increased every second in Groups 3 and 4, whereas it decreased in Groups 1 and 2. The CCCs for each pair of groups in Groups 1&#x2013;4 were 0.8872, 0.9998, 0.8581, and 0.8436, respectively. These results indicate that the synchrony index tended to be higher in Groups 3 and 4, whose competitive activity was higher than that of Groups 1 and 2.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s6">
<title>6 Discussion</title>
<p>The preliminary experiments showed that sensor data from the countdown phase, which had less motion, can be used to identify the differences between conformity and competition training sessions. Hand and head rotations contributed more significantly than position and gaze direction. The activity during the countdown phase in the joint Simon session showed two patterns: one in which the ratio of competitive activities increased during the countdown phase, and another in which it decreased, with the former characterized by greater synchrony. This result contradicts the prediction that synchrony occurs in conformity activities. Therefore, in the main experiment, we added a joint Simon task as a &#x201c;cooperation&#x201d; target for training and created three categories: cooperation, conformity, and competition.</p>
<p>The bot conditions used in the main experiment were created by monitoring the behavior during the motion phase. Therefore, although the countdown phase was involved in the preliminary experiment, a classification category was created in the main experiment using the motion phase. The preliminary experiments showed that hands that were not used for button touching were more important for classification and that they gradually increased or decreased during the countdown up to the motion phase. Based on these results, we expect the features of the classification model using the countdown phase to appear in the classification model using the motion phase.</p>
<sec id="s6-1">
<title>6.1 Main experiment</title>
<sec id="s6-1-1">
<title>6.1.1 Decision tree</title>
<p>Details of machine learning, angular transformations, and statistical models are described in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>.</p>
<p>The ROC value calculated from the true-positive and false-negative rates exceeded 0.98, which was sufficient for the classification accuracy. The confusion matrix and ROC curves are presented in the <xref ref-type="sec" rid="s16">Supplementary Appendix</xref>. The final number of branching nodes was 191. The classification criteria for the top six branches of the decision tree are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The classification criteria for the decision tree. Note: Branching conditions up to the third level are shown in the <xref ref-type="sec" rid="s16">Supplementary Appendix</xref>. The decision tree shows branching conditions based on VR sensor measurements. Numerical values represent threshold values for splitting nodes. Each condition evaluates whether the sensor measurement is below (&#x3c;) the specified threshold, with left branches representing true conditions and right branches representing false conditions. Terminal nodes indicate the final classification: cm (competition), rjo (joint simon), syn (conformity). Variable abbreviations: Hrightrot (Host right-hand rotation), Hleftrot (Host left-hand rotation), Hhead (Host head position), Cleftrot (Client left-hand rotation), Chead (Client head position), Hright (Host right-hand position), Cleft (Client left-hand position).</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g004.tif">
<alt-text content-type="machine-generated">Decision tree diagram for classification criteria with nodes labeled Hrightrot, Hlefrot, Hhead, Hright, Cleft, and Chead, featuring numerical thresholds. Leaf nodes indicate classifications as &#x22;syn&#x22;, &#x22;cm&#x22;, or &#x22;rjo&#x22; in different colored boxes.</alt-text>
</graphic>
</fig>
<p>The confusion matrix is presented in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Confusion matrix of predicted target and actual values from cross-validation. Note: Cross-validation was performed using 10% of the observed values of Session 3, 4, and 5 as test data. True-positive and true-negative rates are shown on the right. The ROC value calculated from these values was 0.98.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g005.tif">
<alt-text content-type="machine-generated">Confusion matrix shows predicted versus true class for &#x22;competition,&#x22; &#x22;joint simon,&#x22; and &#x22;conformity.&#x22; Correct predictions are 86.4%, 95.5%, 95.7% respectively, highlighted in blue. Incorrect predictions are highlighted in pink. Performance metrics table shows TPR and FNR: competition (86.4%, 13.6%), joint simon (95.5%, 4.5%), conformity (95.7%, 4.3%). ROC is 0.98.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s6-1-2">
<title>6.1.2 Effects of human or bot conditions</title>
<p>Avatar behavior varied systematically between conditions. In the human condition, avatars reflected participants&#x2019; real-time movements via the VR tracking system, providing natural responsiveness to each participant&#x2019;s actions. In the bot condition, avatars replayed pre-recorded human movements from earlier experimental sessions, with no adaptive responses to participant behavior. All bot avatar motions were pre-recorded in all bot sessions (Sessions 6 and 7), and there was no dynamic adjustment to participant behavior.</p>
<p>Details of machine learning, angular transformations, and statistical models are described in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>.</p>
<p>These three activity indices (cooperation, conformity, and competition) exhibit mutually constrained relationships. The correlation between concordance and competition was uncorrelated in both conditions, whereas the correlation between cooperation and competition indicators was <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>.5432 (<inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0198) in the human condition and <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>.5078 (<inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0314) in the bot condition. The correlation between the cooperation and conformity indices was <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>.7821 (<inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3c; 0.001) in the human condition and <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; <inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>.6061 (<inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0077) in the bot condition, both of which were high. Therefore, to examine the effect of the conditions on the three activities, we examined each dependent variable individually. <xref ref-type="fig" rid="F6">Figure 6</xref> shows the angular-transformed mean values for each condition before centralization by the group mean.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Effect of the human and bot condition on paired activity. Note: The vertical axis represents the mean classification probability after angular transformation, with values ranging from a minimum of 16.54 to a maximum of 78.69. The error bars indicate 95% confidence intervals.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g006.tif">
<alt-text content-type="machine-generated">Line graph showing activity levels for cooperation, conformity, and competition under human and bot conditions. Cooperation decreases from 50 to 30, conformity increases slightly, and competition remains steady. Error bars indicate variability.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> shows the angular-transformed mean values for cooperation, conformity, and competition activity indices under human and bot conditions. Cooperation activity was significantly higher in the human condition, whereas conformity activity was significantly higher in the bot condition.</p>
<p>An analysis of the human or bot-condition effect with the cooperation indicator as the dependent variable revealed that the estimated value of 7.37 (<inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 3.25) was significant (<inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>(9) &#x3d; 2.27, <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0495). Similarly, a comparison of the mean estimate of the neighborhood with the human and bot conditions of 1 and 0, respectively, was significant (<inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 2.22; Holm&#x2019;s <inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0262). As shown in <xref ref-type="fig" rid="F6">Figure 6</xref>, the ratio of cooperation activity in the human condition was higher than that in the bot condition. An examination of the human- or bot-condition effect with the conformity activity as the dependent variable showed that the estimated value of <inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>6.56 (<inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 2.24) was significant (<inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>(9) &#x3d; 2.93, <inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0167) and that the difference in the marginal mean estimate was substantial (<inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 2.93; Holm&#x2019;s <inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0034), i.e., statistically significant. As shown in <xref ref-type="fig" rid="F6">Figure 6</xref>, the conformity activity in the bot condition was higher than that in the human condition. An examination of the human-/bot-condition effect with competition activity as the dependent variable showed that the estimated value for the condition effect was insignificant, i.e., <inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.43 (<inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 2.28). No difference in competitive activity was observed; however, cooperation activity was significantly greater in the human condition, whereas conformity activity was significantly greater in the bot condition.</p>
<p>An analysis of the human- and bot-condition effects with the JSE as the dependent variable revealed that the estimated value of 0.0033 (<inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.009) was insignificant. Meanwhile, an examination of the human- and bot-condition effects with the percentage of correct responses to the joint Simon task as the dependent variable found that the estimated value of <inline-formula id="inf37">
<mml:math id="m37">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.0111 (<inline-formula id="inf38">
<mml:math id="m38">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0043) was significant (<inline-formula id="inf39">
<mml:math id="m39">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>(8.55) &#x3d; 2.61, <inline-formula id="inf40">
<mml:math id="m40">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0294). Although a trend toward a higher percentage of correct responses was observed in the bot condition, a test of the difference between the marginal estimates showed <inline-formula id="inf41">
<mml:math id="m41">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 1.03 (<inline-formula id="inf42">
<mml:math id="m42">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0535), which was not significant. The correct response rates are presented in <xref ref-type="fig" rid="F7">Figure 7</xref>. The variance in the percentage of correct responses was higher in the human condition, whereas that in the bot condition was minimal. This is presumably because the bots consistently provided correct answers to the questions. In the bot condition, bots replayed pre-recorded human movements taken from correct trials, resulting in consistently correct responses for every trial.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Correct response rates for each condition. Note: The vertical axis represents the correct response rate, and the error bars indicate 95% confidence intervals. The vertical axis represents the average percentage of correct responses during the human and bot sessions. Box plots and distributions are shown on the right. Green and red indicate human and bot conditions, respectively.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g007.tif">
<alt-text content-type="machine-generated">Dot plot comparing percentage of correct answers between human and bot conditions. Human data points cluster near one hundred percent, while bot data points show more variability. Violin plot shows distribution differences.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> presents the percent correct responses for the joint Simon task in each condition using a raincloud plot, which shows the distributions and box plots for each condition.</p>
<p>An analysis of the effect of the human or bot conditions on the synchrony index as the dependent variable showed the estimated value was 0.274 (<inline-formula id="inf43">
<mml:math id="m43">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0035), which was significant (<inline-formula id="inf44">
<mml:math id="m44">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>(7.73) &#x3d; 3.65, <inline-formula id="inf45">
<mml:math id="m45">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3c; 0.001). Additionally, the difference in the marginal estimates was significant (<inline-formula id="inf46">
<mml:math id="m46">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 6.68, <inline-formula id="inf47">
<mml:math id="m47">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3c; 0.001). The results for each group are illustrated in <xref ref-type="fig" rid="F8">Figure 8</xref>, where lower means and higher variances for synchrony were indicated under the bot condition. The lack of synchrony with the bot may have caused this difference, depending on whether the pair was aware or unaware of the bot. However, the difference in the mean bot awareness (0,1) was insignificant. Next, we performed mediation analysis based on the bot condition to determine whether bot awareness was a mediating variable.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Raincloud plot of synchrony. Note: Green and red indicate human and bot conditions, respectively. The vertical axis represents the synchrony index for human and bot sessions. The maximum value was set to 1. Box plots and distributions are shown on the right. Green and red indicate human and bot conditions, respectively.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g008.tif">
<alt-text content-type="machine-generated">Graph showing cross-correlation coefficients for human and bot conditions. Green dots represent human data clustered around a coefficient of 1.0. Orange dots for bots show a wider range. Box plot with violin plot on the right illustrates distribution differences.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s6-1-3">
<title>6.1.3 Mediation analysis</title>
<p>Mediation analysis was conducted separately to investigate the effects of activity on the joint Simon task performance under the human and bot conditions. The variables considered were the three activities and synchrony indices as predictor variables, bot cognition as a mediating variable, and the JSE and correct response rate as the outcome variables. A 32-trial average was considered for the activity and synchrony indices to align with the correct response rate and sample size.</p>
<p>Owing to the high correlation between the three activity indicators that served as predictor variables, we performed principal component analysis as a standard procedure to avoid multiple linearities. Two principal components were extracted when eigenvalues greater than 1 were specified. The factor-loading matrix without rotation is presented in the <xref ref-type="sec" rid="s16">Supplementary Appendix</xref>. Because the first principal component separated cooperation from other activities, we named the factor score of the first principal component the collaboration factor, i.e., collabH (as shown in <xref ref-type="fig" rid="F9">Figure 9</xref>) and as collabB (as shown in <xref ref-type="fig" rid="F10">Figure 10</xref>) for the bot condition. As the second principal component distinguished between competition and conformity, the score for the second principal component factor was named the competition factor, as indicated by competeH and competeB in <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F10">10</xref>, respectively. The first principal component was converted from negative to positive values, with higher values indicating greater cooperation. The outcome variables, JSE, and correct response rate were standardized and entered, and the regression coefficients were reported as standardized coefficients.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Path plot for the human condition. Note: The vertical axis represents standardized values of the dependent variables (JSE and correct response rate), while the horizontal axis represents the predictor variables (collaboration and competition factors). The numbers presented for each path are standardized path coefficients, where collabH and competeH denote the first and second principal components, respectively. Bot is a dummy variable with yes &#x3d; 1 and no &#x3d; 0. JSE and correct response rates are standardized.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g009.tif">
<alt-text content-type="machine-generated">Flowchart with nodes clH, cmH, XcH, NBO, JSH, and CrH connected by arrows labeled with numerical values. Each node has a definition on the side: JSH (JointSimonH), CrH (CorrectH), NBO (NoticeBOT), clH (collabH), cmH (competeH), and XcH (XcorrH).</alt-text>
</graphic>
</fig>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Path plot for the bot condition. Note: The vertical axis represents standardized values of the dependent variables (JSE and correct response rate), while the horizontal axis represents the predictor variables (collaboration and competition factors). collabX and competeX denote PCA components; bot cognition is a binary dummy; all dependent variables are standardized. The numbers shown for each path are standardized path coefficients; collabB and competeB indicate the first and second principal components, respectively.</p>
</caption>
<graphic xlink:href="frvir-06-1623764-g010.tif">
<alt-text content-type="machine-generated">Diagram showing relationships among six variables: JointSimonB (JSB), CorrectB (CrB), NoticeBOT (NBO), collabB (clB), competeB (cmB), and XcorrB (XcB). Arrows represent influence directions, with values indicating effect strength.</alt-text>
</graphic>
</fig>
<p>The path coefficients from the independent variables (two activity factors and synchrony) to the dependent variable (correct response rate) under the human condition are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>. The significant paths revealed that the collaboration factor increased the correct response rate, whereas the competition factor decreased the correct response rate. The statistics for each path are presented in <xref ref-type="table" rid="T2">Table 2</xref> of the <xref ref-type="sec" rid="s16">Supplementary Appendix</xref>. No effect on the JSE was observed, and bot cognition was not shown to be a mediating variable. The total <inline-formula id="inf48">
<mml:math id="m48">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> values for the paths to the JSE, correct response rate, and bot cognition were 0.06, 0.28, and 0.11, respectively.</p>
<p>The path coefficients for the same variables under the bot condition are shown in <xref ref-type="fig" rid="F10">Figure 10</xref>. The total <inline-formula id="inf49">
<mml:math id="m49">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>R</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> values for the paths to the JSE, correct response rate, and bot cognition were 0.37, 0.10, and 0.35, respectively. As a key pathway, collaboration factors had a substantial total effect on enhancing bot cognition and reducing the JSE (<inline-formula id="inf50">
<mml:math id="m50">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; <inline-formula id="inf51">
<mml:math id="m51">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>3.17, <inline-formula id="inf52">
<mml:math id="m52">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0015).</p>
<p>To further examine the robustness of the mediation effect, we replaced the binary &#x201c;Bot-Notice&#x201d; variable with the continuous Social Presence index described in the methods section. However, when using Social Presence as a mediator, we did not observe any significant indirect effects on either the JSE or correct response rate in either condition.</p>
<p>The path analysis results for both conditions are presented in <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F10">10</xref> (see <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F10">10</xref>). These figures illustrate the standardized path coefficients and demonstrate the differential effects of collaboration and competition factors on task performance under human and bot conditions.</p>
<p>The total <italic>R</italic>
<sup>2</sup> values for the paths to the JSE, correct response rate, and bot cognition were 0.06, 0.28, and 0.11, respectively, for the human condition. For the bot condition, the total <italic>R</italic>
<sup>2</sup> values were 0.37, 0.10, and 0.35, respectively. As a significant path, the collaboration factor substantially increased bot cognition and weakened the JSE as a total effect (<inline-formula id="inf53">
<mml:math id="m53">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; <inline-formula id="inf54">
<mml:math id="m54">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>3.17, <inline-formula id="inf55">
<mml:math id="m55">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.0015).</p>
<p>We confirmed that an avatar&#x2019;s appearance did not affect the JSE or bot cognition. Specifically, avatar differences under the avatar condition (Sessions 6 and 7) were compared based on the classification probability as repeated factors. The results of an ANOVA indicated that the main effect of the avatar condition and the interaction effect between the avatar condition and classification were insignificant. Moreover, no significant interaction effects involving the bot avatar appearance were indicated.</p>
<p>In the preliminary experiment, a two-category model comprising competition and conformity was established initially. However, because the participants showed higher synchrony in competitive activities, the model was refined into a three-category classification comprising cooperation, conformity, and competition in the main experiment. The results of the main experiment revealed the feasibility of classifying joint activities based on subtle movements during Phase 3, i.e., when the participants were in motion, and in Phase 1, i.e., when the participants remained still. These results corroborate the predictions of the preliminary experiment, which identified the potential for preparing joint activities during the countdown phase. Furthermore, the results above suggest that the classification model and synchrony index used in this study were valid. A notable finding was the consistent selection of similar features for the synchrony index, which emerged as the most crucial feature for classification in both the preliminary and main experiments. This feature, which is a rotation of the unused hand, would not be readily observed by oneself or others, which suggests that behavioral synchronization phenomena appear as unconscious responses.</p>
<p>In the main experiment, we hypothesized that synchrony and cooperative activity under the bot condition would decrease compared with the human condition. A linear mixed model was used to analyze the effects of human and bot conditions on joint activities and synchrony indices. The results revealed a higher ratio of cooperative activity in the human condition and a high ratio of conformity in the bot condition. This finding is consistent with previous research indicating that humans tend to conform more to non-human agents when the agent&#x2019;s behavior is predictable or lacks social cues. The synchrony index was significantly lower in the bot condition, indicating reduced interpersonal synchrony with bot avatars. This suggests that while bots can elicit conformity, they may not facilitate the same level of behavioral coordination as human partners.</p>
<p>The mediation analysis further revealed that cooperation factors had a substantial total effect on enhancing bot cognition and reducing the JSE. However, bot cognition was not shown to be a mediating variable for the correct response rate or JSE. These findings highlight the complexity of human interaction with a non-responsive avatar and suggest that while participants may recognize non-responsive avatars as joint-action task partners, this recognition does not necessarily translate into improved task performance or increased synchrony.</p>
<p>These findings contribute to discussions on the mechanisms underlying the Joint Simon Effect (JSE). Rather than supporting one theoretical account over the other, our results indicate that bodily synchrony engages both processes simultaneously: participants appeared to use the partner as a spatial reference point while also sharing aspects of task representation. This suggests that interpersonal synchrony can serve as a behavioral signature of these intertwined mechanisms, highlighting how spatial coding and co-representation may co-occur rather than operate in isolation.</p>
</sec>
</sec>
</sec>
<sec id="s7">
<title>7 Limitations and future directions</title>
<p>This study has several limitations. First, the sample size was relatively small, which may limit the generalizability of the findings. Future studies should include larger and more diverse participant groups to validate the classification model and synchrony index. Second, the non-responsive avatars in this study were based on replayed human motion data and did not exhibit adaptive or interactive behaviors. Incorporating more sophisticated AI-driven avatars that can respond dynamically to human actions may yield different results and provide deeper insights into human interaction with responsive avatars. Third, the experimental tasks were limited to a specific joint Simon paradigm in a controlled VR environment. Expanding the range of tasks and exploring real-world applications will be important for understanding the broader applicability of these methods.</p>
<p>In terms of social presence, our findings replicated those of <xref ref-type="bibr" rid="B23">Munnukka et al. (2022)</xref>, in that the visual appearance of the avatar did not significantly affect perceived anthropomorphism or social presence. However, the social presence index derived from questionnaire items did not significantly mediate any of the observed effects. This may be due to the limited sample size, which constrained both the statistical power and the number of items included in the questionnaire. Specifically, we used only two items to assess perceived realism and human-likeness of the avatar. In ongoing studies, we are addressing this limitation by incorporating a more comprehensive set of items to better capture the multidimensional nature of social presence.</p>
<p>Future research should also investigate the neural and psychological mechanisms underlying interpersonal synchrony and joint task performance in VR, as well as the impact of different types of avatars on social interaction. By addressing these limitations and exploring new directions, future studies can further advance our understanding of human-machine interaction in virtual environments.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: The data behind this analysis has been made publicly available at OPENICPSR and can be accessed at (<ext-link ext-link-type="uri" xlink:href="https://www.openicpsr.org/openicpsr/project/178601/version/V1/view">https://www.openicpsr.org/openicpsr/project/178601/version/V1/view</ext-link>). The VR sensor log data are available from the corresponding author upon request.</p>
</sec>
<sec sec-type="ethics-statement" id="s9">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the Ethics Committee of Kyoto University of Advanced Science (Project No. 22H07). The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s10">
<title>Author contributions</title>
<p>YA: Writing &#x2013; review and editing, Funding acquisition, Supervision, Investigation, Software, Conceptualization, Writing &#x2013; original draft, Resources, Project administration, Validation, Methodology, Visualization, Formal Analysis, Data curation. YH: Conceptualization, Validation, Investigation, Writing &#x2013; review and editing, Supervision, Methodology, Software, Data curation, Visualization. MO: Formal Analysis, Conceptualization, Methodology, Data curation, Investigation, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s11">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This study was supported by JSPS KAKENHI Grant Number 21K02988 and 23K22343, 23K25727.</p>
</sec>
<ack>
<p>We thank all participants for their time and effort in this study. We also thank the members of the Center for Social and Psychological Research of Metaverse for their valuable discussions and feedback.</p>
</ack>
<sec sec-type="COI-statement" id="s12">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s13">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s15">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s16">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frvir.2025.1623764/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frvir.2025.1623764/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Development and validation of a robot social presence measurement dimension scale</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>1502</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-28561-8</pub-id>
<pub-id pub-id-type="pmid">36707628</pub-id>
</citation>
</ref>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Decety</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Michalska</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Kinzler</surname>
<given-names>K. D.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>The contribution of emotion and cognition to moral sensitivity: a neurodevelopmental study</article-title>. <source>Cereb. Cortex</source> <volume>22</volume>, <fpage>209</fpage>&#x2013;<lpage>220</lpage>. <pub-id pub-id-type="doi">10.1093/cercor/bhr111</pub-id>
<pub-id pub-id-type="pmid">21616985</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dolk</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hommel</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Prinz</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liepelt</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>The (not so) social simon effect: a referential coding account</article-title>. <source>J. Exp. Psychol. Hum. Percept. Perform.</source> <volume>40</volume>, <fpage>1248</fpage>&#x2013;<lpage>1260</lpage>. <pub-id pub-id-type="doi">10.1037/a0031031</pub-id>
<pub-id pub-id-type="pmid">23339346</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fronda</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Balconi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The effects of prosocial and antisocial behaviors on brain-to-brain synchrony during cooperative tasks</article-title>. <source>Soc. Neurosci.</source> <volume>17</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1080/17470919.2021.1970012</pub-id>
<pub-id pub-id-type="pmid">35045797</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guastello</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Reiter</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Malon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Timm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Shircel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shaline</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Catastrophe theory for dynamical systems in psychology</article-title>. <source>Nonlinear Dyn. Psychol. Life Sci.</source> <volume>27</volume>, <fpage>1</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1891/NDP-2023-0001</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Group identity modulates brain-to-brain synchrony and cooperative decision-making</article-title>. <source>Soc. Cognitive Affect. Neurosci.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1093/scan/nsad123</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harada</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Arima</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Okada</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Effect of virtual interactions through avatar agents on the joint simon effect</article-title>. <source>PLOS ONE</source> <volume>20</volume>, <fpage>e0317091</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0317091</pub-id>
<pub-id pub-id-type="pmid">39792901</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heider</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Simmel</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1944</year>). <article-title>An experimental study of apparent behavior</article-title>. <source>Am. J. Psychol.</source> <volume>57</volume>, <fpage>243</fpage>&#x2013;<lpage>259</lpage>. <pub-id pub-id-type="doi">10.2307/1416950</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liepelt</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Stenzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lappe</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The role of the anterior cingulate cortex in the joint simon effect</article-title>. <source>Front. Psychol.</source> <volume>7</volume>, <fpage>1862</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2016.01862</pub-id>
<pub-id pub-id-type="pmid">27933028</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miss</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pfeuffer</surname>
<given-names>C. U.</given-names>
</name>
<name>
<surname>Kunde</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The joint simon effect depends on perceived agency, but not intentionality, of the alternative action</article-title>. <source>Psychol. Res.</source> <volume>86</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1007/s00426-020-01460-8</pub-id>
<pub-id pub-id-type="pmid">33604724</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Munnukka</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Talvitie-Lamberg</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Maity</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Anthropomorphism and social presence in human&#x2010;virtual service assistant interactions: the role of dialog length and attitudes</article-title>. <source>Comput. Hum. Behav.</source> <volume>135</volume>, <fpage>107343</fpage>. <pub-id pub-id-type="doi">10.1016/j.chb.2022.107343</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paladino</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Mazzurega</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pavani</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Schubert</surname>
<given-names>T. W.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Synchronous multisensory stimulation blurs self-other boundaries</article-title>. <source>Psychol. Sci.</source> <volume>21</volume>, <fpage>1202</fpage>&#x2013;<lpage>1207</lpage>. <pub-id pub-id-type="doi">10.1177/0956797610379234</pub-id>
<pub-id pub-id-type="pmid">20679523</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rennung</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>G&#xf6;ritz</surname>
<given-names>A. S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Taking turns or not? Children&#x2019;s prosocial responsiveness in dyadic and triadic interactions</article-title>. <source>J. Exp. Child Psychol.</source> <volume>141</volume>, <fpage>299</fpage>&#x2013;<lpage>309</lpage>. <pub-id pub-id-type="doi">10.1016/j.jecp.2015.07.009</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sangati</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Willemse</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hunnius</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The role of action prediction and inhibitory control in the joint simon effect</article-title>. <source>Psychol. Res.</source> <volume>85</volume>, <fpage>1001</fpage>&#x2013;<lpage>1013</lpage>. <pub-id pub-id-type="doi">10.1007/s00426-020-01305-4</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sebanz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Knoblich</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Prinz</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Representing others&#x2019; actions: just like one&#x2019;s own?</article-title> <source>Cognition</source> <volume>88</volume>, <fpage>B11</fpage>&#x2013;<lpage>B21</lpage>. <pub-id pub-id-type="doi">10.1016/S0010-0277(03)00043-X</pub-id>
<pub-id pub-id-type="pmid">12804818</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sekitani</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Murakami</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Framework for comparing accuracy of time-series forecasting methods</article-title>. In: <source>2022 International Congress on Advanced Applied Informatics (IIAI-AAI)</source>. <publisher-loc>Kanazawa, Japan</publisher-loc>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sellaro</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Dolk</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Colzato</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Liepelt</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hommel</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Referential coding does not rely on location features: evidence for a non-spatial joint simon effect</article-title>. <source>J. Exp. Psychol. Hum. Percept. Perform.</source> <volume>41</volume>, <fpage>186</fpage>&#x2013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.1037/a0038548</pub-id>
<pub-id pub-id-type="pmid">25528013</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simon</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>1969</year>). <article-title>Reactions toward the source of stimulation</article-title>. <source>J. Exp. Psychol.</source> <volume>81</volume>, <fpage>174</fpage>&#x2013;<lpage>176</lpage>. <pub-id pub-id-type="doi">10.1037/h0027448</pub-id>
<pub-id pub-id-type="pmid">5812172</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smykovskyi</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Koval</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kostiuk</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Emotional contagion and interpersonal synchrony in virtual reality</article-title>. <source>Front. Psychol.</source> <volume>15</volume>, <fpage>1234567</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2024.1234567</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sogemeier</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Naujoks</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Forster</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Krems</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Keinath</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Exploring the interaction between anthropomorphism and performance on trust and acceptance in in-vehicle voice assistants</article-title>. <source>Preprint. SSRN</source>. <pub-id pub-id-type="doi">10.2139/ssrn.4637565</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stenzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liepelt</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>The moving rubber hand illusion revisited: comparing movements and visuotactile stimulation to induce illusory ownership</article-title>. <source>Conscious. Cognition</source> <volume>26</volume>, <fpage>117</fpage>&#x2013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1016/j.concog.2014.02.003</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stenzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chinellato</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bou</surname>
<given-names>M. A. T.</given-names>
</name>
<name>
<surname>del Pobil</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Lappe</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liepelt</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>When humanoid robots become human-like interaction partners: corepresentation of robotic actions</article-title>. <source>J. Exp. Psychol. Hum. Percept. Perform.</source> <volume>38</volume>, <fpage>1073</fpage>&#x2013;<lpage>1077</lpage>. <pub-id pub-id-type="doi">10.1037/a0029493</pub-id>
<pub-id pub-id-type="pmid">22866762</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stenzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dolk</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hommel</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liepelt</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The joint simon effect: a review and theoretical integration</article-title>. <source>Front. Psychol.</source> <volume>7</volume>, <fpage>975</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2016.00975</pub-id>
<pub-id pub-id-type="pmid">27445935</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Kuo</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Hung</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Tzeng</surname>
<given-names>O. J. L.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>A common coding framework in self-other interaction: evidence from joint action task</article-title>. <source>Exp. Brain Res.</source> <volume>182</volume>, <fpage>41</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1007/s00221-007-0972-6</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Kuo</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Hung</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Tzeng</surname>
<given-names>O. J. L.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Action co-representation is tuned to other humans</article-title>. <source>J. Cognitive Neurosci.</source> <volume>20</volume>, <fpage>2015</fpage>&#x2013;<lpage>2024</lpage>. <pub-id pub-id-type="doi">10.1162/jocn.2008.20144</pub-id>
<pub-id pub-id-type="pmid">18416679</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>