<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychiatry</journal-id>
<journal-title>Frontiers in Psychiatry</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychiatry</abbrev-journal-title>
<issn pub-type="epub">1664-0640</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyt.2025.1656292</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Psychiatry</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Predicting affective engagement and mental strain from prosodic speech features</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yache</surname>
<given-names>Vaishnavi Prakash</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2921910/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Moradbakhti</surname>
<given-names>Laura</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2921383/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Neuner</surname>
<given-names>Irene</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Veselinovic</surname>
<given-names>Tanja</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/969684/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Digital Mental Health Lab, Psychiatry, Psychotherapy and Psychosomatic, RWTH Aachen</institution>, <addr-line>Aachen</addr-line>,&#xa0;<country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Institute of Neuroscience and Medicine - 4, Forschungszentrum J&#xfc;lich</institution>, <addr-line>J&#xfc;lich</addr-line>,&#xa0;<country>Germany</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Psychiatry and Psychotherapy II, LVR-Hospital Cologne</institution>, <addr-line>Cologne</addr-line>,&#xa0;<country>Germany</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1364632/overview">Panagiotis Tzirakis</ext-link>, Hume AI, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/82558/overview">Yosuke Morishima</ext-link>, University of Bern, Switzerland</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/291588/overview">Eva Pettemeridou</ext-link>, University of Limassol, Cyprus</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Vaishnavi Prakash Yache, <email xlink:href="mailto:vyache@ukaachen.de">vyache@ukaachen.de</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1656292</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Yache, Moradbakhti, Neuner and Veselinovic.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Yache, Moradbakhti, Neuner and Veselinovic</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Emotional resilience (traditionally defined as the capacity to recover from adversity) and cognitive load (the mental effort for processing information) are critical aspects of mental health functioning. Traditional assessment methods, such as physiological sensors and post-task surveys, often disrupt natural behavior and fail to provide real-time insights. Speech prosody, encompassing pitch, intensity, loudness, and voice activity, offer a non-intrusive alternative for evaluating these psychological constructs. However, the relationship between speech prosody, emotional resilience, and cognitive load remains underexplored, particularly in conversational contexts.</p>
</sec>
<sec>
<title>Objective</title>
<p>This study proposes proxy measures for these constructs based on self-reported engagement, enjoyment, boredom, and cognitive effort during dyadic conversation. By leveraging the SEWA (Automatic Sentiment Estimation in the Wild) database, developed through a European research project on emotion recognition, the research seeks to develop machine learning models that correlate speech patterns with subjective self-reports of emotional and cognitive states.</p>
</sec>
<sec>
<title>Methods</title>
<p>Prosodic features, such as pitch variation, vocal intensity, and voice activity, were extracted from the SEWA database recordings. These features are then normalized to account for inter-speaker variability and used as predictors in machine learning models. Regression and classification models are employed to correlate speech features with subjective self-reports, which serve as ground truth for Positive Affective Engagement (as a proxy for emotional resilience) and Perceived Mental Strain (as a proxy for cognitive load). Data from English and German speakers are analyzed separately to account for linguistic and cultural differences.</p>
</sec>
<sec>
<title>Outcomes</title>
<p>The study establishes a significant relationship between speech prosody and psychological states, demonstrating that Positive Affective Engagement (as a proxy for emotional resilience) and Perceived Mental Strain (as a proxy for cognitive load) can be effectively predicted through prosodic features. Higher emotional resilience is linked to more discernible prosodic patterns in German speech, such as higher loudness and greater voice probability consistency. In contrast, cognitive load prediction remains consistent across English and German datasets.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>This research introduces a novel approach for assessing Positive Affective Engagement (as a proxy for emotional resilience) and Perceived Mental Strain (as a proxy for cognitive load) through speech prosody, highlighting the significant impact of language-specific variations. By combining prosodic features with machine learning techniques, the study offers a promising alternative to traditional psychological assessments. The findings emphasize the need for tailored, multilingual models to accurately estimate psychological states, with potential applications in mental health monitoring, cognitive workload analysis, and human-computer interaction. This work lays the foundation for future innovations in speech-based psychological profiling, advancing our understanding of human emotional and cognitive states in diverse linguistic contexts.</p>
</sec>
</abstract>
<kwd-group>
<kwd>speech prosody</kwd>
<kwd>positive affective engagement</kwd>
<kwd>perceived mental strain</kwd>
<kwd>machine learning in mental health</kwd>
<kwd>prosodic feature extraction</kwd>
<kwd>human-computer interaction</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="6"/>
<equation-count count="0"/>
<ref-count count="53"/>
<page-count count="15"/>
<word-count count="7694"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Digital Mental Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Emotional resilience is defined in the psychological literature as the capacity to adapt to and recover from adversity and stress (<xref ref-type="bibr" rid="B1">1</xref>). It influences significantly how individuals cope with challenges, regulate emotions, and sustain psychological health in demanding circumstances (<xref ref-type="bibr" rid="B2">2</xref>). Emotionally resilient individuals often exhibit behaviors such as sustained engagement, emotional regulation, and cognitive flexibility. These qualities have been studied through self-reports, behavioral observations, and increasingly through physiological or vocal markers (<xref ref-type="bibr" rid="B3">3</xref>). In the context of speech, certain vocal characteristics may be indicative of resilient emotional states. (<xref ref-type="bibr" rid="B4">4</xref>) demonstrated the possibility of estimating personal resilience from speech and physiological signals. Thereby, the most resilience-relevant features were spectral features, including ones related to the fundamental frequency, auditory spectrum coefficients, Mel Frequency Cepstral Coefficients, spectral slope, spectral flux and spectral harmonicity. Further, (<xref ref-type="bibr" rid="B5">5</xref>) were able to demonstrate physiological distress, as opposed to emotional resilience by analyzing 24 vocal characteristics with a machine learning approach. Similarly, one other group successfully used audio-based markers from free speech responses one month post-trauma to accurately classify major depressive disorder (MDD) and post-traumatic stress disorder (PTSD), demonstrating the potential of vocal biomarkers for early mental health diagnosis following traumatic events (<xref ref-type="bibr" rid="B6">6</xref>). For instance, consistent vocal prosody, adaptive modulation of pitch and intensity, and sustained voice activity can reflect stable emotional engagement and regulation, which are hallmarks of emotional resilience.</p>
<p>Similarly, cognitive load, traditionally refers to the mental effort required to process and retain information, plays a fundamental role in determining learning efficiency, productivity, and task performance (<xref ref-type="bibr" rid="B7">7</xref>). High cognitive load can impair decision-making, hinder performance, and induce stress, while balanced cognitive load demands to promote engagement and effective problem-solving (<xref ref-type="bibr" rid="B8">8</xref>). According to Cognitive Load Theory (CLT), load can be intrinsic (task complexity), extraneous (task presentation), or germane (learning effort). High cognitive load is typically associated with slower speech rates, more hesitations, and decreased prosodic variation (<xref ref-type="bibr" rid="B9">9</xref>). Cognitive load and resilience also show complex interrelations. Positive association was demonstrated between resilience and both, intrinsic and extraneous cognitive load, (<xref ref-type="bibr" rid="B10">10</xref>)]. Also, people with high resilience exhibited better global cognitive status and reduced risk of cognitive impairment, even during stressful or demanding periods (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>) On the other side, elevated cognitive load correlates with poor mental health (<xref ref-type="bibr" rid="B13">13</xref>).</p>
<p>Despite their significance, assessing emotional resilience and cognitive load remains challenging. Traditional methods such as physiological monitoring (e.g. heart rate variability, galvanic skin response) and post-task self-reports are widely used but often intrusive, expensive, and impractical in real-world settings (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B16">16</xref>). These techniques interfere with natural behavior and fail to capture real-time fluctuations in emotional and cognitive states. As a result, researchers are increasingly exploring non-intrusive alternatives to measure these psychological constructs in everyday interactions.</p>
<p>Speech is a fundamental mode of human communication and an emerging source of psychological insight. It provides a rich, real-time signal that reflects cognitive and emotional states without disrupting natural interactions (<xref ref-type="bibr" rid="B17">17</xref>). Variations in speech prosody, such as pitch, rhythm, intensity, and pause patterns, are directly influenced by underlying psychological conditions (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B19">19</xref>). For example, individuals under high cognitive load tend to exhibit slower speech rates, increased hesitation, and prolonged pauses due to elevated mental effort (<xref ref-type="bibr" rid="B20">20</xref>, <xref ref-type="bibr" rid="B21">21</xref>). Similarly, emotionally resilient individuals may maintain stable pitch variation and consistent speech intensity, reflecting better emotional regulation and adaptability.</p>
<p>Research highlights the growing potential of speech as a non-invasive biomarker for mental health conditions such as depression. The acoustic and temporal characteristics of speech have been shown to correlate strongly with depressive states, enabling automated screening and monitoring in clinical and real world settings (<xref ref-type="bibr" rid="B22">22</xref>). For example, (<xref ref-type="bibr" rid="B23">23</xref>) demonstrated that the prosodic and spectral features extracted from the speech effectively distinguish depressed individuals from controls, confirming the diagnostic value of the speech. Similarly, (<xref ref-type="bibr" rid="B24">24</xref>) and (<xref ref-type="bibr" rid="B25">25</xref>) employed speech recognition technology to analyze timing-related features, such as speech rate and pauses, finding significant associations with depression severity. Furthermore, deep learning approaches have recently improved detection accuracy by modeling complex vocal patterns, as evidenced by (<xref ref-type="bibr" rid="B26">26</xref>), who developed neural architectures capable of capturing subtle speech characteristics related to depression. These advances support the integration of speech-based assessments into scalable, real-time mental health monitoring platforms.</p>
<p>Previous studies have explored speech-based emotion recognition and cognitive load estimation (<xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B27">27</xref>), and prosodic features have been increasingly used to detect neurological and psychological conditions such as depression and schizophrenia (<xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B30">30</xref>). Moreover, systematic reviews highlight the growing potential of voice analysis for detecting neurological and mood disorders, emphasizing how emerging artificial intelligence (AI) techniques can uncover objective markers of mental health conditions from speech signals (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>). These findings reinforce the promise of non-invasive, speech-based approaches for psychological assessment, particularly for mental health monitoring in both clinical and everyday settings.</p>
<p>Moreover, recent advancements have incorporated semantic information into speech emotion recognition frameworks. By combining semantic and paralinguistic features, models can capture both the content and the expressive nuances of speech, leading to improved performance in emotion detection tasks (<xref ref-type="bibr" rid="B33">33</xref>). Multimodal approaches that integrate textual, acoustic, and visual modalities have demonstrated superior accuracy, particularly in capturing complex affective states during natural interactions (<xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B35">35</xref>). These frameworks often employ deep learning architectures such as transformers or recurrent networks to model temporal dependencies and contextual cues, enhancing emotion inference over isolated prosodic features alone (<xref ref-type="bibr" rid="B36">36</xref>).</p>
<p>Despite these advancements, the interplay between emotional resilience, cognitive load, and prosody remains underexplored, particularly in conversational settings. Understanding this relationship could lead to the development of automated tools for real-time psychological assessment, benefiting mental health diagnostics, educational support systems, and human-computer interaction.</p>
<p>Furthermore, prosodic markers are influenced by linguistic and cultural factors, raising concerns about generalizing speech-based models across different languages. Even though some machine learning models for speech-based detection of neurological and/or psychological disorders already include data from multiple languages (<xref ref-type="bibr" rid="B29">29</xref>), the majority of studies focus on one language only (<xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B30">30</xref>). This highlights the importance of accounting for language-specific prosodic patterns when developing speech-based detection models, to ensure both reliability and fairness across diverse populations.</p>
<p>These prosodic attributes are then used as predictors in machine learning models to classify or predict subjective self-reports. Techniques such as regression models (for continuous prediction of psychological states) and classification models (for high vs. low resilience or cognitive load) are applied. Additionally, to account for inter-speaker differences, features are normalized, and models are trained separately for English and German speakers, considering linguistic and cultural differences in prosody (<xref ref-type="bibr" rid="B28">28</xref>).</p>
<p>This research introduces a novel framework for non-intrusive psychological assessment through voice analysis. The key contributions include:</p>
<list list-type="bullet">
<list-item>
<p>Conversational context analysis, unlike traditional speech emotion recognition, this study examines interpersonal dynamics (e.g., agreement, engagement) and their impact on resilience and cognitive load.</p>
</list-item>
<list-item>
<p>Non-Intrusive psychological profiling, by eliminating the need for physiological sensors or intrusive self-reports, offering a real-time, speech-only approach.</p>
</list-item>
<list-item>
<p>Cross-language considerations, by developing models that account for linguistic and cultural differences in prosody, providing broader applicability.</p>
</list-item>
</list>
<p>Potential applications range from mental health monitoring (e.g., stress and resilience assessment) to real-time cognitive support in education and workplace settings. Furthermore, human-computer interaction systems, such as virtual assistants, could benefit from adaptive responses based on users&#x2019; emotional and cognitive states.</p>
<p>By integrating speech prosody with self-reported emotional resilience and cognitive load measures, this research advances our understanding of how voice reflects psychological states. The proposed machine learning framework paves the way for automated, real-time assessment tools that enhance mental health monitoring, learning environments, and human-machine interactions. Through a deeper exploration of prosody&#x2019;s role in emotional and cognitive processes, this study contributes to the ongoing evolution of voice-based psychological profiling.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Methodology</title>
<sec id="s2_1">
<label>2.1</label>
<title>Dataset and participants</title>
<p>This study utilizes the SEWA <italic>Sentiment and Emotion in the Wild</italic> dataset, a rich collection of dyadic conversations where participants discuss emotionally evocative advertisements (<xref ref-type="bibr" rid="B37">37</xref>). These interactions inherently involve cognitive effort (processing the advertisements) and emotional exchange (responding to a conversation partner), making them an ideal context for examining speech-based indicators of emotional resilience and cognitive load. The SEWA database also provides subjective self-reports, in which participants rate their engagement, emotional arousal, and conversational experience on a scale of -5 to 5. These self-reports serve as ground truth labels for machine learning models.</p>
<p>To facilitate individual-level speech analysis, the conversations were separated into individual speaker recordings, ensuring that each participant&#x2019;s speech was analyzed independently. For segmentation, the Hugging Face library was used, which provides tools to efficiently process and separate the individual speaker recordings from the dyadic conversations (<xref ref-type="bibr" rid="B38">38</xref>). This allowed for independent analysis of each participant&#x2019;s speech recordings, facilitating accurate emotional and cognitive state assessment.</p>
<p>After segmentation, the dataset consisted of 66 native English-speaking and 64 native German speaking participants, totaling 130 individual recordings. Each participant provided self-reports on their emotional and cognitive states, which served as the ground truth for model training.</p>
<p>The average age of participants in both datasets is relatively similar, with English speakers averaging approximately 34.94 years and German speakers 31.08 years. The gender distribution is balanced in the English dataset (50% male, 50% female) but shows a slight male majority in the German dataset (60.94% male, 39.06% female).</p>
<p>Although the SEWA is publicly available for academic use, access was obtained through formal request to the dataset organizers, and usage adhered to all stated terms and conditions. All recordings are anonymized, and the dataset includes consent from participants for secondary research, ensuring compliance with ethical standards for human data use.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Feature extraction</title>
<p>To assess Positive Affective Engagement (emotional resilience) and Perceived Mental Strain (cognitive load) through speech prosody, multiple acoustic features were extracted from the segmented speech data. Feature extraction was performed using the openSMILE toolkit (<xref ref-type="bibr" rid="B17">17</xref>), a widely used, open-source toolkit developed for the extraction of audio features from speech and music signals, which provides robust prosodic feature analysis.</p>
<p>It is particularly renowned for its efficiency in processing large datasets and its applicability in real-time systems. The toolkit provides a comprehensive set of features, including prosodic elements such as pitch, loudness, and voice quality, which are essential for analyzing emotional states in speech (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>). In the context of emotion recognition, prosodic features extracted using openSMILE have been instrumental in capturing the nuances of emotional expression. For instance, studies have utilized openSMILE to extract features like fundamental frequency (F0), intensity, and voice quality measures, which are then analyzed to infer emotional states. These features are computed over short time frames and can be aggregated to provide both local and global perspectives on the speaker&#x2019;s emotional state (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B41">41</xref>, <xref ref-type="bibr" rid="B42">42</xref>).</p>
<p>The prosodic features computed are listed in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Speech feature categories and their descriptions (<xref ref-type="bibr" rid="B43">43</xref>, <xref ref-type="bibr" rid="B44">44</xref>, <xref ref-type="bibr" rid="B45">45</xref>).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Category</th>
<th valign="middle" align="left">Features (Abbreviations)</th>
<th valign="middle" align="left">Description</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Fundamental Frequency (Pitch)</td>
<td valign="middle" align="left">F0_sma_de_amean_mean, F0_sma_de_skewness_mean</td>
<td valign="middle" align="left">Reflects vocal fold vibration and is linked to emotional engagement.</td>
</tr>
<tr>
<td valign="middle" align="left">Intensity &amp; Loudness</td>
<td valign="middle" align="left">Pcm_intensity_sma_amean_mean, pcm_intensity_sma_de_amean_mean, pcm_loudness_sma_amean_mean,<break/>Pcm_loudness_sma_de_amean_mean</td>
<td valign="middle" align="left">Measures vocal energy, associated with emotional arousal.</td>
</tr>
<tr>
<td valign="middle" align="left">Spectral Features (MFCCs)</td>
<td valign="middle" align="left">mfcc_sma_de[1]_skewness_mean, mfcc_sma_de[2]_skewness_mean</td>
<td valign="middle" align="left">Mel-frequency cepstral coefficients<break/>(MFCCs) capture timbre and vocal tone, which vary with cognitive and emotional states.</td>
</tr>
<tr>
<td valign="middle" align="left">Temporal &amp; Voice Activity</td>
<td valign="middle" align="left">Pcm_zcr_sma_amean_mean, voiceProb_sma_amean_mean</td>
<td valign="middle" align="left">Zero-crossing rate (ZCR) measures the frequency of signal sign changes, reflecting rhythm and articulation rate, while voice probability<break/>(VoiceProb) estimates the likelihood of speech being voiced rather than silent or unvoiced, offering insights into vocal activity and mental effort.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We selected a focused set of acoustic features (e.g., F0, intensity, MFCCs, temporal and voice activity) based on prior literature indicating their relevance to emotional and cognitive states (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B46">46</xref>). This approach balances interpretability and reduces the risk of overfitting, particularly when working with moderate-sized datasets.</p>
<p>All extracted features were normalized to account for inter-speaker variability, ensuring consistency across different voices and linguistic backgrounds.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Annotation and ground truth labels</title>
<p>The SEWA dataset includes subjective self-reports from participants, which were used as proxies, ground truth labels for model training and evaluation of psychological states.</p>
<p>Participants rated their emotional and cognitive experiences on a scale of -5 to 5 across dimensions.</p>
<p>While these do not match formal definitions in clinical psychology, we operationalized, these self-reports to create two target variables,:</p>
<list list-type="bullet">
<list-item>
<p>Positive Affective Engagement (as a proxy for Emotional Resilience): As the aggregation of participant&#x2019;s affective positivity and engagement during the interaction (engagement + enjoyment + positive feelings), reflecting affective adaptability within the conversational context. While not equivalent to clinical definitions of emotional resilience &#x2014; which require adaptation to adversity &#x2014; this score served as a proxy for momentary emotional adaptability in a socially interactive context.</p>
</list-item>
<list-item>
<p>Perceived Mental Strain (as a proxy for Cognitive Load): This approach reflects the extent of mental strain or discomfort reported during the conversation (negative feelings &#x2212; engagement &#x2212; enjoyment). However, we acknowledge that this proxy does not align directly with traditional definitions of cognitive load, which emphasize task complexity and working memory demands (<xref ref-type="bibr" rid="B47">47</xref>). Thus, our cognitive load metric should be interpreted as a subjective impression of effortful or aversive cognitive experience.</p>
</list-item>
</list>
<p>The processed labels were then integrated into the dataset alongside the extracted prosodic features, forming the input for machine learning models. By leveraging both binary and multi-class categorization, the study ensured flexibility in predictive modeling, allowing for both high-level classification and nuanced regression analysis. This approach not only strengthened the interpretability of the models but also facilitated a more comprehensive understanding of how speech prosody correlates with cognitive and emotional resilience in real-world conversations.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Machine learning model development</title>
<p>To assess Positive Affective Engagement (emotional resilience) and Perceived Mental Strain (cognitive load) from speech prosody, we implemented a machine learning pipeline using Support Vector Machines (SVM) for classification and linear regression for continuous score prediction. This approach aligns with established methodologies in affective computing and speech-based psychological assessment, where SVMs have been widely used due to their robustness in high-dimensional feature spaces and ability to handle non-linear patterns (<xref ref-type="bibr" rid="B27">27</xref>). The methodology consists of feature preprocessing, dimensionality reduction, classification, and evaluation.</p>
<p>The models were developed with the following workflow:</p>
<list list-type="order">
<list-item>
<p>Feature Preprocessing and Dimensionality Reduction: Speech prosodic features extracted from the SEWA dataset were standardized using the StandardScaler to mitigate inter-speaker variability. Principal Component Analysis (PCA) was then applied to reduce dimensionality while preserving the most informative variance in the data. The top three principal components were retained as feature representations for subsequent modeling.</p>
</list-item>
<list-item>
<p>Data Merging and Categorization: Self-reported emotional Positive Affective Engagement (emotional resilience) and Perceived Mental Strain (cognitive load) scores were used as ground truth. The self-reports were divided into high and low categories based on quantile thresholds. Scores above the 66th percentile were categorized as high, while those below the 66th percentile were labeled as low. The categorized data was merged with the PCA-transformed features, creating a structured dataset for classification and regression analysis. This approach was chosen to preserve a larger portion of the dataset for analysis while still creating distinguishable classes. Compared to stricter splits, which reduce the dataset to 66% of its original size, the 66th percentile method allows better data utilization and model generalization. Additionally, this strategy employs a moderate threshold to strike a balance between ensuring adequate class separation and retaining sufficient data for reliable model training and evaluation (<xref ref-type="bibr" rid="B48">48</xref>, <xref ref-type="bibr" rid="B49">49</xref>).</p>
</list-item>
<list-item>
<p>Classification and Regression Models: For classification tasks, we trained SVM models to predict binary labels for emotional resilience and cognitive load. The dataset was split into training (80%) and testing (20%) sets, ensuring stratification for balanced class representation. Feature normalization was reapplied to maintain consistency across training and testing data. Additionally, linear regression models were trained to predict continuous self-report scores, allowing for a more granular assessment of psychological attributes.</p>
</list-item>
<list-item>
<p>Model Evaluation: Performance metrics, including accuracy, precision, recall, and F1-score, were computed for classification models, while regression performance was assessed using standard error metrics. The models were trained and validated separately for English and German datasets to account for linguistic variations in prosody. To account for variability in model performance due to limited sample size, we applied bootstrapping with 1,000 iterations for both regression and classification tasks. In each iteration, the data were resampled with replacement, followed by model training and evaluation on a stratified test split. This allowed us to compute 95% confidence intervals for key metrics (e.g., accuracy, precision, F1-score), providing a more robust estimate of generalization performance than a single train-test split. Bootstrapping was particularly valuable for assessing the stability and reliability of model predictions across language groups. The results demonstrated the feasibility of using speech features to infer emotional resilience and cognitive load, highlighting the potential of non-intrusive psychological assessment through voice analysis.</p>
</list-item>
</list>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Cross-language analysis</title>
<p>Given the linguistic and cultural differences in prosody, separate models were trained and validated for English and German speakers. This ensured that variations in speech patterns due to language differences did not bias the results. The comparative analysis between the two language groups provided insights into the universality of prosodic indicators for psychological assessment. By isolating language groups, the models more accurately capture prosodic markers relevant within each linguistic context, improving prediction robustness.</p>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Ethical considerations</title>
<p>The study adhered to ethical guidelines for working with human speech data. The SEWA data set was used in accordance with its licensing agreements, ensuring participant anonymity and privacy. Since the study involves secondary data analysis, no direct interaction with participants was required. The dataset includes participant consent for secondary research and is fully anonymized. As no identifiable information was used and no new data were collected, additional ethics approval was not required according to standard practices for secondary anonymized data analysis.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<p>The primary aim of this study was to investigate the relationship between speech prosody and psychological constructs like Positive Affective Engagement and Perceived Mental Strain, using a combination of subjective self-reports and prosodic features from the SEWA database. The results presented in this section provide empirical evidence of how speech characteristics such as pitch, intensity, loudness, spectral features and voice activity can serve as indicators of an individual&#x2019;s emotional resilience and cognitive load.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Descriptive statistics of participant demographics</title>
<p>The study utilizes a subset of the SEWA dataset, which includes dyadic conversations in English and German. The dataset consists of 66 participants in the English subset and 64 participants in the German subset, as summarized in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison of English and German datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Category</th>
<th valign="middle" align="center">English dataset</th>
<th valign="middle" align="center">German dataset</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Number of Participants</td>
<td valign="middle" align="center">66</td>
<td valign="middle" align="center">64</td>
</tr>
<tr>
<td valign="middle" align="left">Average Age</td>
<td valign="middle" align="center">34.94</td>
<td valign="middle" align="center">31.08</td>
</tr>
<tr>
<td valign="middle" align="left">Gender Distribution</td>
<td valign="middle" align="center">50% male, 50% female</td>
<td valign="middle" align="center">60.94% male, 39.06% female</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The histograms in <xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1</bold>
</xref>, <xref ref-type="fig" rid="f2">
<bold>2</bold>
</xref> illustrate the distributions of Positive Affective Engagement (emotional resilience) and Perceived Mental Strain (cognitive load) scores for English and Germanspeaking participants, respectively. These distributions provide insights into how individuals from different linguistic backgrounds perceive and report their psychological states.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Distribution of ground truth scores for Positive Affective Engagement (left) and Perceived Mental Strain (right) in the English dataset. Histograms show participant self-reports, with kernel density estimates overlaid to illustrate score distributions.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1656292-g001.tif">
<alt-text content-type="machine-generated">Side-by-side histograms with overlaid density plots. The left chart shows the distribution of emotional resilience scores, and the right chart displays cognitive load scores. Both have a range from negative ten to fifteen on the x-axis and count on the y-axis.</alt-text>
</graphic>
</fig>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Distribution of ground truth scores for Positive Affective Engagement (left) and Perceived Mental Strain (right) in the German dataset. Histograms show participant self-reports, with kernel density estimates overlaid to illustrate score distributions.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1656292-g002.tif">
<alt-text content-type="machine-generated">Histogram on the left shows the distribution of emotional resilience scores in German, peaking around five. The right histogram displays cognitive load scores with a peak around zero. Both include a normal distribution curve overlay.</alt-text>
</graphic>
</fig>
<p>Both groups display multimodal distributions, indicating diverse experiences of Positive Affective Engagement (emotional resilience). English speakers tend to report slightly higher resilience on average, with a more balanced spread between positive and negative values. German speakers show a concentration of scores in the negative range, potentially indicating a more critical self-assessment or cultural differences in reporting resilience. German participants report slightly higher Perceived Mental Strain (cognitive load) on average, with fewer instances of extremely low values. English participants show a more evenly distributed pattern, including both high and low cognitive load responses. The stronger peak around 5 in the German dataset suggests a potential cultural or linguistic difference in task perception or self-reporting tendencies. These differences might stem from cultural factors, language-specific prosodic variations, or differing interpretations of the rating scales.</p>
<p>This analysis provides a foundational understanding of how participants self-assess emotional and cognitive states, setting the stage for further statistical comparisons and machine learning modeling.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Feature extraction and visualization</title>
<p>To understand the relationship between speech prosody and psychological states, various acoustic features were extracted (using OpenSMILE) and analyzed for both English and German speakers. <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> summarizes the mean and standard deviation of key prosodic features, categorized into fundamental frequency, intensity &amp; loudness, spectral features, and temporal &amp; voice activity parameters.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Mean and standard deviation of prosodic features.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Feature category</th>
<th valign="middle" align="left">Feature name</th>
<th valign="middle" align="center">English</th>
<th valign="middle" align="center">German</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" colspan="4" align="left">Fundamental frequency</th>
</tr>
<tr>
<td valign="middle" align="left">F0 (Hz)</td>
<td valign="middle" align="left">F0_sma_de_amean_mean</td>
<td valign="middle" align="left">-0.1895 &#xb1; 0.1830</td>
<td valign="middle" align="left">0.0129 &#xb1; 0.0076</td>
</tr>
<tr>
<td valign="middle" align="left">F0_Skew</td>
<td valign="middle" align="left">F0_sma_de_skewness_mean</td>
<td valign="middle" align="left">0.0102 &#xb1; 0.0038</td>
<td valign="middle" align="left">-0.0079 &#xb1; 0.0084</td>
</tr>
<tr>
<th valign="middle" colspan="4" align="left">Intensity, loudness</th>
</tr>
<tr>
<td valign="middle" align="left">Intensity (mdB)</td>
<td valign="middle" align="left">pcm_intensity_sma_amean_mean</td>
<td valign="middle" align="left">67.8 &#xb1; 195.5</td>
<td valign="middle" align="center">73.6 &#xb1; 170.9</td>
</tr>
<tr>
<td valign="middle" align="left">Intensity_&#x394; (mdB)</td>
<td valign="middle" align="left">pcm_intensity_sma_de_amean_mean</td>
<td valign="middle" align="left">167.4 &#xb1; 262</td>
<td valign="middle" align="center">55.6 &#xb1; 170.2</td>
</tr>
<tr>
<td valign="middle" align="left">Loudness (dB)</td>
<td valign="middle" align="left">pcm_loudness_sma_amean_mean</td>
<td valign="middle" align="left">0.8432 &#xb1; 0.0200</td>
<td valign="middle" align="left">0.9238 &#xb1; 0.0596</td>
</tr>
<tr>
<td valign="middle" align="left">Loudness _&#x394; (mdB)</td>
<td valign="middle" align="left">pcm_loudness_sma_de_amean_mean</td>
<td valign="middle" align="left">19.54 &#xb1; 115.24</td>
<td valign="middle" align="left">507.3 &#xb1; 1571.9</td>
</tr>
<tr>
<th valign="middle" colspan="4" align="left">Spectral features</th>
</tr>
<tr>
<td valign="middle" align="left">MFCC1_Skew</td>
<td valign="middle" align="left">mfcc_sma_de[1]_skewness_mean</td>
<td valign="middle" align="left">0.1848 &#xb1; 0.0380</td>
<td valign="middle" align="left">-0.0564 &#xb1; 0.0475</td>
</tr>
<tr>
<td valign="middle" align="left">MFCC2_Skew</td>
<td valign="middle" align="left">mfcc_sma_de[2]_skewness_mean</td>
<td valign="middle" align="left">-0.3539 &#xb1; 0.0748</td>
<td valign="middle" align="left">-0.2348 &#xb1; 0.0476</td>
</tr>
<tr>
<th valign="middle" colspan="4" align="left">Temporal, voice activity</th>
</tr>
<tr>
<td valign="middle" align="left">ZCR</td>
<td valign="middle" align="left">pcm_zcr_sma_amean_mean</td>
<td valign="middle" align="left">0.0728 &#xb1; 0.0044</td>
<td valign="middle" align="left">0.0583 &#xb1; 0.0107</td>
</tr>
<tr>
<td valign="middle" align="left">VoiceProb</td>
<td valign="middle" align="left">voiceProb_sma_amean_mean</td>
<td valign="middle" align="left">0.5653 &#xb1; 0.0136</td>
<td valign="middle" align="left">0.6329 &#xb1; 0.0090</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>This table compares prosodic feature measurements (mean &#xb1; standard deviation) between English and German speech samples. Abbreviations: F0, Fundamental Frequency; F0 Skew, Skewness of Fundamental Frequency; Intensity, Root Mean Square (RMS) Intensity; Intensity <sub>&#x394;</sub>, Derivative of Intensity; Loudness, Perceived Loudness in Decibels (dB); Loudness <sub>&#x394;</sub>, Derivative of Loudness; MFCC1 Skew, Skewness of the 1st Mel-Frequency Cepstral Coefficient; MFCC2 Skew, Skewness of the 2nd Mel-Frequency Cepstral Coefficient; ZCR, Zero-Crossing Rate; VoiceProb, Probability of Voice Activity. Intensity values are based on openSMILE feature extraction. The unit &#x201c;mdB&#x201d; is used for readability, values should be interpreted as relative intensity.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<list list-type="order">
<list-item>
<p>Fundamental Frequency (Pitch Variability).</p>
<list list-type="bullet">
<list-item>
<p>The F0 mean derivative (F0 sma de amean mean) is slightly negative for English speakers (0.1895&#xa0;Hz) but positive for German speakers (0.0129&#xa0;Hz), suggesting that English speakers exhibit greater pitch fluctuations, which may indicate more dynamic intonation. (This feature represents how rapidly pitch changes on average; larger absolute values suggest greater pitch movement over time.).</p>
</list-item>
<list-item>
<p>The F0 skewness (F0 sma de skewness mean) is positive for English speakers (0.0102) but negative for German speakers (-0.0079), implying that English speech may be more varied in tone, while German speech has a more balanced pitch distribution. (Skewness reflects asymmetry in the pitch distribution&#x2014;positive values indicate a longer tail on the right, suggesting more high-pitched variations.).</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Intensity &amp; Loudness.</p>
<list list-type="bullet">
<list-item>
<p>The mean intensity (pcm intensity sma amean mean) is higher in German speakers (73.6 mdB) than in English speakers (67.8 mdB), indicating that German speech tends to be louder overall. (Intensity corresponds to the energy or perceived volume of the signal, measured in decibels.).</p>
</list-item>
<list-item>
<p>The intensity derivative (pcm intensity sma de amean mean) shows a larger fluctuation for English speakers (167.4 mdB) compared to German speakers (55.6 mdB), suggesting that English conversations exhibit more dynamic loudness variations. (The derivative indicates how quickly loudness changes over time&#x2014;larger values reflect greater loudness modulation.).</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Spectral Features (MFCC Analysis).</p>
<list list-type="bullet">
<list-item>
<p>The skewness of the first Mel-Frequency Cepstral Coefficient (mfcc sma de[1] skewness mean) is positive for English speakers (0.1848) but negative for German speakers (-0.0564), indicating different spectral energy distributions. (MFCC1 captures coarse spectral shape; its skewness reveals whether the energy distribution leans toward higher or lower frequencies.) (<xref ref-type="bibr" rid="B50">50</xref>).</p>
</list-item>
<list-item>
<p>The second MFCC skewness (mfcc sma de[2] skewness mean) is lower in German speakers (-0.2348) compared to English speakers (-0.3539). (MFCC2 reflects finer spectral details; negative skewness indicates more concentration of energy in lower coefficients, possibly linked to vowel or consonant articulation styles.) (<xref ref-type="bibr" rid="B50">50</xref>).</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Temporal &amp; Voice Activity Features.</p>
<list list-type="bullet">
<list-item>
<p>Zero-crossing rate (pcm zcr sma amean mean) is higher in English speech (0.0728) than in German speech (0.0583), indicating a more frequent transition between voiced and unvoiced speech sounds in English. (The zero-crossing rate reflects how often the audio waveform crosses the zero amplitude line, i.e., switches from positive to negative or vice versa&#x2014;and is typically higher in unvoiced or noisy segments.).</p>
</list-item>
<list-item>
<p>Voice probability (voiceProb sma amean mean) is higher in German speakers (0.6329) than in English speakers (0.5653), suggesting that German speakers maintain continuous speech more consistently than English speakers. (Voice probability estimates the likelihood that speech (vs. silence or noise) is present at each moment in the signal.).</p>
</list-item>
</list>
</list-item>
</list>
<p>
<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3</bold>
</xref>, <xref ref-type="fig" rid="f4">
<bold>4</bold>
</xref> display the Pearson correlation heatmaps of extracted prosodic features in the English and German datasets, respectively. The color scale represents correlation coefficients (r) ranging from -1 (strong negative correlation) to 1 (strong positive correlation), with 0 indicating no correlation. Statistically significant correlations (<italic>p &lt;</italic> 0.05) are marked with highlighted boxes, marking feature relationships that are unlikely to have occurred by chance.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Correlation heatmap of prosodic features in English speech, with Pearson coefficients (&#x2013;1 to 1) shown by color and significant correlations (<italic>p &lt;</italic> 0.05) marked by boxes, highlighting strong associations.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1656292-g003.tif">
<alt-text content-type="machine-generated">Heatmap showing feature correlations with statistical significance in English. Features include F0, F0_Skew, Intensity, Loudness, MFCC1_Skew, MFCC2_Skew, ZCR, and VoiceProb. Correlation values range from -1.0 to 1.0, with a color gradient from blue (negative correlation) to red (positive correlation). Notable correlations: F0_Skew with F0 at -0.60, Loudness with F0 at 0.64, and ZCR with Loudness at -0.74.</alt-text>
</graphic>
</fig>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Correlation heatmap of prosodic features in German speech, with Pearson coefficients (&#x2013;1 to 1) shown by color and significant correlations (<italic>p &lt;</italic> 0.05) marked by boxes, highlighting strong associations.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1656292-g004.tif">
<alt-text content-type="machine-generated">Heatmap showing feature correlations with statistical significance in German. The diagonal displays labels: F0, F0_Skew, Intensity, Intensity_D, Loudness, Loudness_D, MFCC1_Skew, MFCC2_Skew, ZCR, VoiceProb. Positive correlations are in red, negative in blue, ranging from minus one to one. Notable values include 0.75 for F0_Skew with F0 and -0.95 for ZCR with Intensity.</alt-text>
</graphic>
</fig>
<p>Overall, the German dataset exhibits stronger and more numerous significant correlations between features, suggesting a more tightly integrated prosodic structure. Notably, Loudness and ZCR show a remarkably strong negative correlation in German (<italic>r</italic> =&#x2212;0.95, <italic>p &lt;</italic> 0.001), but not in English, indicating substantial inter-feature dependency in German speech. This strong inverse relationship can be explained by their distinct acoustic roles: Loudness reflects the perceived intensity of a sound, while ZCR captures how noisy or erratic the signal is. For instance, vowel sounds are often loud yet smooth, resulting in low ZCR, whereas soft background noise may be quiet but chaotic, yielding high ZCR. As such, the two features don&#x2019;t necessarily increase together&#x2014;particularly in German, where clearer, vowel-rich articulation may produce speech that is simultaneously louder and less noisy, reinforcing this negative correlation.</p>
<p>Additional differences are observed in the relationships between spectral features (MFCC1 Skew, MFCC2 Skew) and pitch-based metrics (F0 Skew, Intensity). For example, MFCC1 Skew is positively correlated with F0 Skew (<italic>r</italic> =&#xa0;0.56, <italic>p &lt;</italic> 0.05) and Intensity &#x394; (<italic>r</italic> =&#xa0;0.62, <italic>p &lt;</italic> 0.05) in German but shows weaker and inconsistent correlations in English. These differences underscore language-specific acoustic patterns that likely influence model performance.</p>
<p>Moreover, VoiceProb&#x2014;representing voice activity&#x2014;exhibited moderate correlations with several features in English (e.g., F0, <italic>r</italic> =&#xa0;0.56, <italic>p &lt;</italic> 0.05; Loudness <italic>r</italic> =&#xa0;0.53, <italic>p &lt;</italic> 0.05), while showing weaker and more selective correlations in German, particularly with spectral skew measures (for example, MFCC1 Skew <italic>r</italic> =&#xa0;0.72, <italic>p &lt;</italic> 0.05; MFCC2 Skew(<italic>r</italic> =&#x2212;0.55, <italic>p &lt;</italic> 0.05). This variation suggests that voice activity may be cued differently across languages, influencing how features are weighted during model training and learning.</p>
<p>Principal Component Analysis revealed that the top three components captured a substantial portion of variability in prosodic features: 65.74% for English (PC1: 33.93%, PC2: 18.16%, PC3: 13.65%) and 72.93% for German (PC1: 35.67%, PC2: 25.57%, PC3: 11.69%). The dominant features contributing to PC1 in the English dataset were loudness (0.483), pitch skewness (0.440), zero-crossing rate (0.403), and voice probability (0.392). In the German dataset, PC1 was most influenced by spectral skewness (MFCC1: 0.444, MFCC2: 0.405), pitch skewness (0.435), and intensity dynamics (0.368). These results suggest that PCA retained components with clear links to voice expressivity and speech rhythm, preserving psychological interpretability even after dimensionality reduction.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Assessment of positive affective engagement and perceived mental strain from speech (machine learning)</title>
<p>This section presents the predictive performance of Linear Regression for regression tasks and Support Vector Machine (SVM) for classification. The results highlight differences between the English and German datasets in terms of predicting Emotional Resilience and Cognitive Load.</p>
<p>
<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> displays the Mean Squared Error (MSE) for Positive Affective Engagement and Perceived Mental Strain across the English and German datasets, using both standard regression evaluation and bootstrapped confidence intervals. In the standard evaluation, the German dataset yields a lower MSE for Positive Affective Engagement (27.146) compared to the English dataset (35.583), suggesting a better model fit for German speakers. For Perceived Mental Strain, the MSE is slightly lower for the English dataset (26.01) than for German (28.52), indicating relatively similar predictive performance across languages.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Linear regression results for Positive Affective Engagement and Perceived Mental Strain.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">Metric</th>
<th valign="middle" align="center">English dataset</th>
<th valign="middle" align="center">German dataset</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Positive Affective Engagement</td>
<td valign="middle" align="left">Mean Squared Error (MSE)</td>
<td valign="middle" align="center">35.583</td>
<td valign="middle" align="center">27.146</td>
</tr>
<tr>
<td valign="middle" align="left"/>
<td valign="middle" align="left">Bootstrapped MSE (95% CI)</td>
<td valign="middle" align="center">41.76 [17.63, 92.90]</td>
<td valign="middle" align="center">27.39 [10.54, 51.68]</td>
</tr>
<tr>
<td valign="middle" align="left">Perceived Mental Strain</td>
<td valign="middle" align="left">Mean Squared Error (MSE)</td>
<td valign="middle" align="center">26.01</td>
<td valign="middle" align="center">28.52</td>
</tr>
<tr>
<td valign="middle" align="left"/>
<td valign="middle" align="left">Bootstrapped MSE (95% CI)</td>
<td valign="middle" align="center">36.75 [15.16, 74.23]</td>
<td valign="middle" align="center">23.74 [6.85, 47.11]</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The bootstrapped results further illustrate the uncertainty around these estimates. For Positive Affective Engagement, the German model achieves a bootstrapped MSE of 27.39 with a narrower 95% confidence interval [10.54, 51.68], whereas the English model shows a higher bootstrapped MSE of 41.76 and a wider confidence interval [17.63, 92.90], reflecting greater variability. Similarly, for Perceived Mental Strain, the bootstrapped MSE is lower in the German dataset (23.74; 95% CI: [6.85, 47.11]) compared to the English dataset (36.75; 95% CI: [15.16, 74.23]). These findings reinforce that Positive Affective Engagement is predicted more reliably in the German dataset, while Perceived Mental Strain shows greater uncertainty in the English dataset. The tighter confidence intervals in the German data may indicate more consistent prosodic cues or less variation in self-reporting among German speakers.</p>
<p>
<xref ref-type="table" rid="T5">
<bold>Tables&#xa0;5</bold>
</xref>, <xref ref-type="table" rid="T6">
<bold>6</bold>
</xref> report the classification performance for Positive Affective Engagement and Perceived Mental Strain across English and German datasets. Both standard evaluation metrics (based on a single train-test split) and bootstrapped results (with 95% confidence intervals) are presented for a more robust and reliable assessment of model generalization.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Classification performance for Positive Affective Engagement in English and German datasets (standard and bootstrapped).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Metric</th>
<th valign="middle" colspan="2" align="center">English</th>
<th valign="middle" colspan="2" align="center">German</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left"/>
<td valign="middle" align="center">Standard</td>
<td valign="middle" align="center">Bootstrapped</td>
<td valign="middle" align="center">Standard</td>
<td valign="middle" align="center">Bootstrapped</td>
</tr>
<tr>
<td valign="middle" align="left">Accuracy</td>
<td valign="middle" align="center">0.429</td>
<td valign="middle" align="center">0.622 [0.357&#x2013;0.857]</td>
<td valign="middle" align="center">0.615</td>
<td valign="middle" align="center">0.621 [0.308&#x2013;0.846]</td>
</tr>
<tr>
<td valign="middle" align="left">Macro Precision</td>
<td valign="middle" align="center">0.378</td>
<td valign="middle" align="center">0.611 [0.275&#x2013;0.909]</td>
<td valign="middle" align="center">0.608</td>
<td valign="middle" align="center">0.623 [0.292&#x2013;0.900]</td>
</tr>
<tr>
<td valign="middle" align="left">Macro Recall</td>
<td valign="middle" align="center">0.377</td>
<td valign="middle" align="center">0.597 [0.325&#x2013;0.854]</td>
<td valign="middle" align="center">0.613</td>
<td valign="middle" align="center">0.616 [0.333&#x2013;0.875]</td>
</tr>
<tr>
<td valign="middle" align="left">Macro F1-score</td>
<td valign="middle" align="center">0.378</td>
<td valign="middle" align="center">0.579 [0.300&#x2013;0.845]</td>
<td valign="middle" align="center">0.607</td>
<td valign="middle" align="center">0.590 [0.291&#x2013;0.845]</td>
</tr>
<tr>
<td valign="middle" align="left">Confusion Matrix</td>
<td valign="middle" align="center">[5, 4][4, 1]</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">[5, 3][2, 3]</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Classification performance for Perceived Mental Strain in English and German datasets (standard and bootstrapped).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Metric</th>
<th valign="middle" colspan="2" align="center">English</th>
<th valign="middle" colspan="2" align="center">German</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left"/>
<td valign="middle" align="center">Standard</td>
<td valign="middle" align="center">Bootstrapped</td>
<td valign="middle" align="center">Standard</td>
<td valign="middle" align="center">Bootstrapped</td>
</tr>
<tr>
<td valign="middle" align="left">Accuracy</td>
<td valign="middle" align="center">0.571</td>
<td valign="middle" align="center">0.585 [0.286&#x2013;0.857]</td>
<td valign="middle" align="center">0.538</td>
<td valign="middle" align="center">0.609 [0.308&#x2013;0.846]</td>
</tr>
<tr>
<td valign="middle" align="left">Macro Precision</td>
<td valign="middle" align="center">0.542</td>
<td valign="middle" align="center">0.581 [0.250&#x2013;0.875]</td>
<td valign="middle" align="center">0.548</td>
<td valign="middle" align="center">0.607 [0.278&#x2013;0.889]</td>
</tr>
<tr>
<td valign="middle" align="left">Macro Recall</td>
<td valign="middle" align="center">0.521</td>
<td valign="middle" align="center">0.570 [0.312&#x2013;0.833]</td>
<td valign="middle" align="center">0.550</td>
<td valign="middle" align="center">0.592 [0.325&#x2013;0.845]</td>
</tr>
<tr>
<td valign="middle" align="left">Macro F1-score</td>
<td valign="middle" align="center">0.475</td>
<td valign="middle" align="center">0.542 [0.271&#x2013;0.825]</td>
<td valign="middle" align="center">0.535</td>
<td valign="middle" align="center">0.569 [0.291&#x2013;0.838]</td>
</tr>
<tr>
<td valign="middle" align="left">Confusion Matrix</td>
<td valign="middle" align="center">[7, 1][5, 1]</td>
<td valign="middle" align="center">&#x2013;</td>
<td valign="middle" align="center">[4, 4][2, 3]</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>For Positive Affective Engagement Classification, the German dataset consistently outperforms the English dataset across all metrics, with notably higher Accuracy (0.615 vs. 0.428), Macro Precision (0.608 vs. 0.378), and Macro F1-score (0.607 vs. 0.378). These results suggest that the classifier was more effective in distinguishing between high and low positive affective engagement in the German speech data, possibly due to language-dependent acoustic patterns or cultural response tendencies. Bootstrapped results confirm this trend but reveal greater uncertainty, particularly in the English dataset. For example, the bootstrapped Accuracy for English is 0.622 with a wide confidence interval [0.357&#x2013;0.857], compared to 0.621 [0.308&#x2013;0.846] for German. Despite similar means, the English model exhibits a broader range, indicating less stable generalization. German bootstrapped Macro F1-score (0.590 [0.291&#x2013;0.845]) also edges out the English value (0.579 [0.300&#x2013;0.845]), reinforcing the model&#x2019;s slightly better reliability on German speech.</p>
<p>For Perceived Mental Strain, classification performance is more balanced between languages. The English dataset achieves marginally higher standard Accuracy (0.571 vs. 0.538), while the German dataset outperforms in Macro Recall (0.55 vs. 0.521) and Macro F1-score (0.535 vs. 0.475). Bootstrapped results echo this pattern: English Accuracy averages 0.585 [0.286&#x2013;0.857], whereas German reaches 0.609 [0.308&#x2013;0.846]. Notably, German bootstrapped precision (0.607 [0.278&#x2013;0.889]) and F1-score (0.569 [0.291&#x2013;0.838]) again exceed those of English, suggesting better balance between sensitivity and specificity.</p>
<p>In summary, the results indicate that language plays a crucial role in speech-based psychological assessments, with machine learning models demonstrating different levels of performance on English and German datasets. Overall, the inclusion of bootstrapping highlights the importance of evaluating model stability under data resampling, especially with modest sample sizes. Positive Affective Engagement classification remains clearly stronger for German, supported by both higher point estimates and tighter confidence intervals. Perceived Mental Strain classification is more variable but shows comparable performance across languages. These results emphasize that language-specific acoustic and reporting factors influence both raw model accuracy and its statistical reliability, underscoring the need for culturally aware and multilingual modeling in speech-based psychological assessment.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>In this study, we operationalized Positive Affective Engagement (emotional resilience) and Perceived Mental Strain (cognitive load) using self-report data from the SEWA dataset. Our measure of emotional resilience was derived from participants&#x2019; self-reported engagement, enjoyment, and positive affect in response to emotionally evocative advertisements. This operationalization does not aim to capture trait-level resilience, as defined in psychological literature, but instead reflects momentary affective engagement within conversational settings. The Perceived Mental Strain is calculated as a composite of self-reported cognitive effort, boredom, and enjoyment. These refinements improve theoretical coherence while acknowledging the limitations of using self-reports as proxies for complex psychological constructs.</p>
<p>The findings of this study highlight the role of speech prosody in psychological assessments, emphasizing how language-specific variations impact predictive modeling of Positive Affective Engagement and Perceived Mental Strain. The observed differences between English and German datasets suggest that prosodic features contribute uniquely to psychological state estimation, necessitating careful consideration in multilingual applications.</p>
<p>The results demonstrate that Positive Affective Engagement is more accurately predicted in German speech using linear regression, whereas Perceived Mental Strain prediction remains consistent across languages. One possible explanation for this difference lies in the prosodic variations between English and German speakers. German speech exhibited higher loudness and greater consistency in voice probability, potentially making Positive Affective Engagement more discernible through acoustic features. The stronger correlations among prosodic features in German further support this notion, indicating a more tightly connected acoustic profile that facilitates model learning. These observations align with findings by (<xref ref-type="bibr" rid="B51">51</xref>), who showed that while German and English share similar prosodic structures, native listeners perceive and weight prosodic cues differently, German listeners being more sensitive to pitch rises and English listeners more to pitch falls. This difference in perceptual sensitivity suggests that prosodic cues are encoded and utilized distinctly across languages, which may contribute to the varying predictive performance seen in our models.</p>
<p>The classification results reinforce this observation, showing that high Positive Affective Engagement is more accurately detected in the German dataset, whereas English data better identifies low Perceived Mental Strain. This suggests that linguistic and cultural factors influence speech patterns associated with psychological states. The findings align with prior research that indicates variations in emotional expression and self-reporting tendencies across languages (<xref ref-type="bibr" rid="B29">29</xref>), affecting the reliability of cross-linguistic speech analysis.</p>
<p>The results underscore the importance of tailoring speech-based assessments to account for linguistic and cultural differences. The observed disparities suggest that speech processing models trained on one language may not generalize well to another, necessitating language-specific adaptations. This is particularly relevant for multilingual clinical applications, where accurate psychological state estimation is crucial. Future implementations should explore adaptive modeling approaches that incorporate linguistic variations into speech-based assessments.</p>
<p>Moreover, the study demonstrates that prosodic features such as pitch variability, loudness, and voice activity provide meaningful indicators of psychological states. These findings could inform the development of more robust emotion recognition systems, improving their reliability in real-world applications such as mental health monitoring and human-computer interaction.</p>
<p>Despite the promising findings, this study has several limitations. First, the dataset is relatively small, with only 130 participants across both language groups. A larger and more diverse sample could enhance the generalizability of the results. Additionally, cultural and contextual factors influencing Positive Affective Engagement and Perceived Mental Strain reporting were not explicitly controlled for, which could have influenced the differences. And, while the OpenSmile toolkit provides over thousands of features, incorporating all of them without prior selection would necessitate additional dimensionality reduction and introduce model complexity that may hinder interpretability. Future work may explore automated feature selection from the full feature set.</p>
<p>While the observed performance differences between English and German participants were attributed to prosodic variation, socio-cultural and attitudinal differences in self-reporting may also contribute to these effects. Previous studies have documented cross-cultural biases in self-assessment, with German participants often exhibiting more conservative or critical self-evaluations compared to English-speaking counterparts. Such biases could impact ground truth labels and therefore model outcomes.</p>
<p>Limitation also lies in the interpretability of the feature space. Although PCA reduced dimensionality effectively and preserved key prosodic patterns, it introduces a layer of abstraction that can obscure direct relationships with psychological constructs. While we report loadings and variance explained to improve transparency, future work should consider interpretable models such as LASSO or random forest regressors that maintain a clear mapping between original features and target variables.</p>
<p>Another limitation is the reliance on self-reported measures for Positive Affective Engagement and Perceived Mental Strain. Subjective assessments may introduce bias, as individuals perceive and report their psychological states differently. Incorporating objective physiological measures, such as heart rate variability or galvanic skin response, could provide a more comprehensive evaluation. Although SEWA ratings are subjective and lack psychometric standardization, our composite scores were informed by theoretical frameworks linking enjoyment and engagement to resilience, and cognitive effort to load. To support construct validity, future studies will incorporate validated scales such as the Brief Resilience Coping Scale (BRCS) and NASA-TLX alongside SEWA ratings for triangulation.</p>
<p>Furthermore, while this study focuses on English and German, the findings may not extend to other languages with distinct prosodic characteristics. Future research should explore additional languages to determine whether similar trends persist and refine cross-linguistic models accordingly. While SVMs and linear regression provide interpretable baselines, we acknowledge their limitations in capturing non-linear and sequential dependencies inherent in prosodic speech patterns. Preliminary work using Random Forests and LSTM-based architectures is underway to explore non-linear interactions and temporal modeling.</p>
<p>Building on the current findings, future research should address several key areas. Expanding the dataset with more participants and diverse demographic backgrounds would improve model robustness. Investigating alternative machine learning approaches, such as deep learning models, could enhance prediction accuracy by capturing complex non-linear relationships between speech prosody and psychological states (<xref ref-type="bibr" rid="B26">26</xref>). Moreover, Integration of global and local prosodic features has been shown to enhance the accuracy of emotion recognition systems. Global features capture overarching statistics like mean and standard deviation of prosodic contours, while local features represent temporal dynamics at finer granularities, such as syllables and words (<xref ref-type="bibr" rid="B52">52</xref>). The current study focuses on global prosodic features. However, to capture within-conversation variation in load and resilience, future work will incorporate temporal modeling of prosodic contours (e.g., pitch trajectories, pause timing) at the utterance level. Tools like Praat or Voice Activity Detection (VAD) will enable segmentation aligned with speech turns, allowing for dynamic load tracking across dialogue.</p>
<p>Additionally, exploring cross-linguistic transfer learning could help mitigate performance gaps between languages. Training models on a diverse set of languages and fine-tuning them for specific linguistic contexts could improve generalizability. Future studies should also consider incorporating multimodal data, such as facial expressions and physiological signals, to create a more holistic assessment framework. Finally, real-world validation of these models in clinical or everyday settings would provide practical insights into their effectiveness. Testing speech-based psychological assessment tools in naturalistic environments could help refine their application for mental health monitoring, cognitive workload analysis, and human-computer interaction.</p>
<p>It is important to note that, in this study, the spoken content was tied to an induced emotional state, as participants discussed advertising videos rather than reflecting on their mood in a general sense. This design enables greater experimental control over affective stimuli, but it may not fully capture the natural variability and authenticity present in spontaneous speech. In contrast, studies such as (<xref ref-type="bibr" rid="B53">53</xref>) focus on self-initiated, spontaneous speech to detect mental health risks such as depression and anxiety, offering richer insights into a person&#x2019;s habitual emotional tone. Induced states may elicit different prosodic patterns compared to spontaneous self-disclosure, which should be considered when interpreting the findings. Nonetheless, this approach further highlights the potential of speech as a non-invasive biomarker for early detection of mental health conditions. From a clinical perspective, identifying reliable vocal indicators&#x2014;even from neutral or task-oriented speech&#x2014;could offer valuable insights into an individual&#x2019;s psychological well-being without requiring explicit discussion of sensitive topics. This would be especially relevant in addressing the persistent stigma surrounding mental health issues, supporting unobtrusive monitoring and timely intervention.</p>
<p>In summary, this study demonstrates the impact of language on speech-based psychological assessments, with German data showing stronger predictability for Positive Affective Engagement (emotional resilience) while Perceived Mental Strain (cognitive load) prediction remains similar across languages. The results highlight the importance of language-specific speech features in machine learning models and underscore the need for tailored approaches in multilingual settings. Future research should address dataset limitations, explore alternative modeling techniques, and validate findings in real-world applications to enhance the effectiveness of speech-based psychological assessments.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>This study explored the relationship between speech prosody and psychological constructs, specifically Positive Affective Engagement (emotional resilience) and Perceived Mental Strain (cognitive load), using English and German speech samples from the SEWA dataset. Through statistical analysis and machine learning models, we demonstrated that prosodic features &#x2014; such as pitch variability, intensity, and spectral characteristics &#x2014; serve as meaningful indicators of psychological states. Our results revealed notable language-based differences in speech characteristics and their correlation with emotional and cognitive attributes.</p>
<p>Machine learning models performed differently across the two languages, with linear regression yielding lower errors for Positive Affective Engagement (emotional resilience) in German, while Perceived Mental Strain (cognitive load) prediction remained relatively similar across datasets. Classification results from SVM indicated that high Positive Affective Engagement was more accurately detected in German, whereas low Perceived Mental Strain was better identified in English. These findings suggest that language-specific acoustic patterns influence the reliability of psychological inferences, emphasizing the need for linguistic adaptations in automated speech-based assessments.</p>
<p>While this study provides valuable insights, certain limitations must be acknowledged. The size of the data set was relatively small, and cultural differences in self-reporting may have influenced ground-truth labels. Future research should explore larger and more diverse datasets, apply deep learning techniques tailored to Positive Affective Engagement and Perceived Mental Strain detection, and investigate the generalization of findings across additional languages. By addressing these challenges, speech-based psychological assessment tools can be refined to enhance their accuracy and applicability in multilingual and cross-cultural contexts.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <uri xlink:href="http://db.sewaproject.eu/">http://db.sewaproject.eu/</uri>.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>VY: Writing &#x2013; original draft, Visualization, Data curation, Formal analysis, Conceptualization, Writing &#x2013; review &amp; editing, Software, Methodology. LM: Visualization, Supervision, Writing &#x2013; review &amp; editing, Conceptualization. IR: Writing &#x2013; review &amp; editing, Methodology, Supervision, Investigation. TV: Conceptualization, Supervision, Funding acquisition, Writing &#x2013; review &amp; editing, Visualization.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This research was funded by the German Federal Ministry of Research, Technology and Space (BMFTR) (grant number 16SV9137) as a part of the FRIEND project.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We would like to express our gratitude to all those who have supported and contributed to the completion of this research. Special thanks to Prof. Dr. Thomas Frodl, director of the Department for Psychiatry, Psychotherapy and Psychosomatics, for providing support and resources that significantly contributed to the success of this study. Additionally, we acknowledge the SEWA database for providing the data that formed the foundation of our analysis. Their comprehensive and well-maintained dataset was crucial to the development of this research. This manuscript will be part of the doctoral thesis (Dr. rer. medic.) of Vaishnavi Prakash Yache at the Faculty of Medicine, RWTH Aachen University and it is part of the work within the FRIEND project, which is funded by the German Federal Ministry of Research, Technology and Space (BMFTR) (grant number 16SV9137).</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bonanno</surname> <given-names>GA</given-names>
</name>
</person-group>. <article-title>Loss, trauma, and human resilience: Have we underestimated the human capacity to thrive after extremely aversive events</article-title>? <source>Am psychol Assoc</source>. (<year>2008</year>) <volume>59</volume>(<issue>1</issue>):<page-range>101&#x2013;13</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1037/1942-9681.S.1.101</pub-id>, PMID: <pub-id pub-id-type="pmid">14736317</pub-id></citation></ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Southwick</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Yehuda</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Resilience definitions, theory, and challenges: interdisciplinary perspectives</article-title>. <source>Eur J Psychotraumatol</source>. (<year>2014</year>) <volume>5</volume>:<elocation-id>25338</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3402/ejpt.v5.25338</pub-id>, PMID: <pub-id pub-id-type="pmid">25317257</pub-id></citation></ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalisch</surname> <given-names>R</given-names>
</name>
<name>
<surname>Baker</surname> <given-names>DG</given-names>
</name>
<name>
<surname>Basten</surname> <given-names>U</given-names>
</name>
<name>
<surname>Boks</surname> <given-names>MP</given-names>
</name>
<name>
<surname>Bonanno</surname> <given-names>GA</given-names>
</name>
<name>
<surname>Brummelman</surname> <given-names>E</given-names>
</name>
<etal/>
</person-group>. <article-title>The resilience framework as a strategy to combat stress-related disorders</article-title>. <source>Nat Hum Behav</source>. (<year>2017</year>) <volume>1</volume>:<page-range>784&#x2013;90</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41562-017-0200-8</pub-id>, PMID: <pub-id pub-id-type="pmid">31024125</pub-id></citation></ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hsu</surname> <given-names>S-M</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>S-H</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>T-R</given-names>
</name>
</person-group>. <article-title>Personal resilience can be well estimated from heart rate variability and paralinguistic features during human&#x2013;robot conversations</article-title>. <source>Sensors</source>. (<year>2021</year>) <volume>21</volume>:<fpage>5844</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s21175844</pub-id>, PMID: <pub-id pub-id-type="pmid">34502736</pub-id></citation></ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iyer</surname> <given-names>R</given-names>
</name>
<name>
<surname>Nedeljkovic</surname> <given-names>M</given-names>
</name>
<name>
<surname>Meyer</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Using vocal characteristics to classify psychological distress in adult helpline callers: retrospective observational study</article-title>. <source>JMIR Formative Res</source>. (<year>2022</year>) <volume>6</volume>:<elocation-id>e42249</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.2196/42249</pub-id>, PMID: <pub-id pub-id-type="pmid">36534456</pub-id></citation></ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schultebraucks</surname> <given-names>K</given-names>
</name>
<name>
<surname>Yadav</surname> <given-names>V</given-names>
</name>
<name>
<surname>Shalev</surname> <given-names>AY</given-names>
</name>
<name>
<surname>Bonanno</surname> <given-names>GA</given-names>
</name>
<name>
<surname>Galatzer-Levy</surname> <given-names>IR</given-names>
</name>
</person-group>. <article-title>Deep learning-based classification of posttraumatic stress disorder and depression following trauma utilizing visual and auditory markers of arousal and mood</article-title>. <source>psychol Med</source>. (<year>2022</year>) <volume>52</volume>:<page-range>957&#x2013;67</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/S0033291720002718</pub-id>, PMID: <pub-id pub-id-type="pmid">32744201</pub-id></citation></ref>
<ref id="B7">
<label>7</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>John Sweller</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Ayres</surname> <given-names>P</given-names>
</name>
</person-group>. <source>Cognitive Load Theory</source>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>Elsevier Academic Press</publisher-name> (<year>2011</year>).</citation></ref>
<ref id="B8">
<label>8</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Paas</surname> <given-names>F</given-names>
</name>
<name>
<surname>Tuovinen</surname> <given-names>JE</given-names>
</name>
<name>
<surname>Tabbers</surname> <given-names>H</given-names>
</name>
<name>
<surname>Van Gerven</surname> <given-names>PW</given-names>
</name>
</person-group>. <article-title>Cognitive load measurement as a means to advance cognitive load theory</article-title>. In: <source>Cognitive Load Theory</source>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>Routledge</publisher-name> (<year>2016</year>). p. <fpage>63</fpage>&#x2013;<lpage>71</lpage>.</citation></ref>
<ref id="B9">
<label>9</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yin</surname> <given-names>B</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>F</given-names>
</name>
<name>
<surname>Ruiz</surname> <given-names>N</given-names>
</name>
<name>
<surname>Ambikairajah</surname> <given-names>E</given-names>
</name>
</person-group>. (<year>2008</year>). <article-title>Speech-based cognitive load monitoring system</article-title>, in: <conf-name>2008 IEEE International Conference on Acoustics, Speech and Signal Processing (IEEE)</conf-name>, <publisher-name>IEEE</publisher-name>. pp. <page-range>2041&#x2013;4</page-range>.</citation></ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Omari</surname> <given-names>H</given-names>
</name>
<name>
<surname>Aljawarneh</surname> <given-names>YM</given-names>
</name>
<name>
<surname>Al-Rawashdeh</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>The relationship between resilience and cognitive load among college level students: a cross-sectional study</article-title>. <source>Crit Public Health</source>. (<year>2025</year>) <volume>35</volume>:<fpage>2504074</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/09581596.2025.2504074</pub-id>
</citation></ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>P</given-names>
</name>
<name>
<surname>Li</surname> <given-names>R</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Longitudinal trajectories of psychological resilience and cognitive impairment among older adults: evidence from a national cohort study</article-title>. <source>medRxiv</source>. (<year>2024</year>) <volume>80</volume>(<issue>6</issue>). doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2024.09.02.24312919</pub-id>, PMID: <pub-id pub-id-type="pmid">39989018</pub-id></citation></ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jung</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>GB</given-names>
</name>
<name>
<surname>Nishimi</surname> <given-names>K</given-names>
</name>
<name>
<surname>Chibnik</surname> <given-names>L</given-names>
</name>
<name>
<surname>Koenen</surname> <given-names>KC</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>HC</given-names>
</name>
</person-group>. <article-title>Association between psychological resilience and cognitive function in older adults: effect modification by inflammatory status</article-title>. <source>Geroscience</source>. (<year>2021</year>) <volume>43</volume>:<page-range>2749&#x2013;60</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11357-021-00406-1</pub-id>, PMID: <pub-id pub-id-type="pmid">34184172</pub-id></citation></ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Byrd-Bredbenner</surname> <given-names>C</given-names>
</name>
<name>
<surname>Eck</surname> <given-names>KM</given-names>
</name>
</person-group>. <article-title>Relationships among executive function, cognitive load, and weight-related behaviors in university students</article-title>. <source>Am J Health Behav</source>. (<year>2020</year>) <volume>44</volume>:<fpage>691</fpage>&#x2013;<lpage>703</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5993/AJHB.44.5.12</pub-id>, PMID: <pub-id pub-id-type="pmid">33121586</pub-id></citation></ref>
<ref id="B14">
<label>14</label>
<citation citation-type="book">
<person-group person-group-type="author">
<collab>Cain</collab>
</person-group>. <source>A Review of the Mental Workload Literature</source>. <publisher-name>Technical report, Defence Research and Development Canada Toronto</publisher-name> (<year>2007</year>).</citation></ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Duffy</surname> <given-names>VG</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Measurement and identification of mental workload during simulated computer tasks with multimodal methods and machine learning</article-title>. <source>Ergonomics</source>. (<year>2020</year>) <volume>63</volume>:<fpage>896</fpage>&#x2013;<lpage>908</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/00140139.2020.1759699</pub-id>, PMID: <pub-id pub-id-type="pmid">32330080</pub-id></citation></ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>H-J</given-names>
</name>
<name>
<surname>Labbaf</surname> <given-names>S</given-names>
</name>
<name>
<surname>Borelli</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Dutt</surname> <given-names>N</given-names>
</name>
<name>
<surname>Rahmani</surname> <given-names>AM</given-names>
</name>
</person-group>. <article-title>Objective stress monitoring based on wearable sensors in everyday settings</article-title>. <source>J Med Eng Technol</source>. (<year>2020</year>) <volume>44</volume>:<page-range>177&#x2013;89</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/03091902.2020.1759707</pub-id>, PMID: <pub-id pub-id-type="pmid">32589065</pub-id></citation></ref>
<ref id="B17">
<label>17</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Eyben</surname> <given-names>F</given-names>
</name>
<name>
<surname>W&#xf6;llmer</surname> <given-names>M</given-names>
</name>
<name>
<surname>Schuller</surname> <given-names>B</given-names>
</name>
</person-group>. (<year>2010</year>). <article-title>opensmile: The munich versatile and fast open-source audio feature extractor</article-title>, in: <conf-name>Proceedings of the 18th ACM International Conference on Multimedia (ACM)</conf-name>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>. pp. <page-range>1459&#x2013;62</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/1873951.1874246</pub-id>
</citation></ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scherer</surname> <given-names>KR</given-names>
</name>
</person-group>. <article-title>Vocal communication of emotion: A review of research paradigms</article-title>. <source>Speech Communication</source>. (<year>2003</year>) <volume>40</volume>:<page-range>227&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0167-6393(02)00084-5</pub-id>
</citation></ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Coutinho</surname> <given-names>E</given-names>
</name>
<name>
<surname>Dibben</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>Psychoacoustic cues to emotion in speech prosody and music</article-title>. <source>Cogn Emotion</source>. (<year>2013</year>) <volume>27</volume>:<page-range>658&#x2013;84</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/02699931.2012.732559</pub-id>, PMID: <pub-id pub-id-type="pmid">23057507</pub-id></citation></ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boyer</surname> <given-names>S</given-names>
</name>
<name>
<surname>Paubel</surname> <given-names>P-V</given-names>
</name>
<name>
<surname>Ruiz</surname> <given-names>R</given-names>
</name>
<name>
<surname>El Yagoubi</surname> <given-names>R</given-names>
</name>
<name>
<surname>Daurat</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Human voice as a measure of mental load level</article-title>. <source>J Speech Language Hearing Res</source>. (<year>2018</year>) <volume>61</volume>:<page-range>2722&#x2013;34</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1044/2018JSLHR-S-18-0066</pub-id>, PMID: <pub-id pub-id-type="pmid">30383160</pub-id></citation></ref>
<ref id="B21">
<label>21</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ouyang</surname> <given-names>A</given-names>
</name>
<name>
<surname>Dang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Sethu</surname> <given-names>V</given-names>
</name>
<name>
<surname>Ambikairajah</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Speech based emotion prediction: Can a linear model work</article-title>? In: <source>
<italic>Proceedings of INTERSPEECH</italic> (ISCA)</source>. <publisher-loc>Graz, Austria</publisher-loc> (<year>2019</year>). p. <page-range>2813&#x2013;7</page-range>.</citation></ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gumus</surname> <given-names>M</given-names>
</name>
<name>
<surname>DeSouza</surname> <given-names>DD</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Fidalgo</surname> <given-names>C</given-names>
</name>
<name>
<surname>Simpson</surname> <given-names>W</given-names>
</name>
<name>
<surname>Robin</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Evaluating the utility of daily speech assessments for monitoring depression symptoms</article-title>. <source>Digital Health</source>. (<year>2023</year>) <volume>9</volume>:<fpage>20552076231180523</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/20552076231180523</pub-id>, PMID: <pub-id pub-id-type="pmid">37426590</pub-id></citation></ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Tao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Speech-based clinical depression screening: An empirical study</article-title>. <source>arXiv preprint arXiv:2406.03510</source>. (<year>2024</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2406.03510</pub-id>
</citation></ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yamamoto</surname> <given-names>M</given-names>
</name>
<name>
<surname>Takamiya</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sawada</surname> <given-names>K</given-names>
</name>
<name>
<surname>Yoshimura</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kitazawa</surname> <given-names>M</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>KC</given-names>
</name>
<etal/>
</person-group>. <article-title>Using speech recognition technology to investigate the association between timing-related speech features and depression severity</article-title>. <source>PloS One</source>. (<year>2020</year>) <volume>15</volume>:<elocation-id>e0238726</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0238726</pub-id>, PMID: <pub-id pub-id-type="pmid">32915846</pub-id></citation></ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trevino</surname> <given-names>AC</given-names>
</name>
<name>
<surname>Quatieri</surname> <given-names>TF</given-names>
</name>
<name>
<surname>Malyska</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>Phonologically-based biomarkers for major depressive disorder</article-title>. <source>EURASIP J Adv Signal Process</source>. (<year>2011</year>) <volume>2011</volume>:<fpage>1</fpage>&#x2013;<lpage>18</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1687-6180-2011-42</pub-id>
</citation></ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Jing</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Deep learning for depression recognition from speech</article-title>. <source>Mobile Networks Appl</source>. (<year>2023</year>) <volume>29</volume>:<page-range>1212&#x2013;27</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11036-022-02086-3</pub-id>
</citation></ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schuller</surname> <given-names>BASS</given-names>
</name>
<name>
<surname>B. and Seppi</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Recognizing realistic emotions and affect in speech: State of the art and lessons learnt from the first challenge</article-title>. <source>Speech Communication</source>. (<year>2011</year>) <volume>53</volume>:<page-range>1062&#x2013;87</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.specom.2011.01.011</pub-id>
</citation></ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cummins</surname> <given-names>N</given-names>
</name>
<name>
<surname>Scherer</surname> <given-names>S</given-names>
</name>
<name>
<surname>Krajewski</surname> <given-names>J</given-names>
</name>
<name>
<surname>Schnieder</surname> <given-names>S</given-names>
</name>
<name>
<surname>Epps</surname> <given-names>J</given-names>
</name>
<name>
<surname>Quatieri</surname> <given-names>TF</given-names>
</name>
</person-group>. <article-title>A review of depression and suicide risk assessment using speech analysis</article-title>. <source>Speech Communication</source>. (<year>2015</year>) <volume>71</volume>:<fpage>10</fpage>&#x2013;<lpage>49</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.specom.2015.03.004</pub-id>
</citation></ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parola</surname> <given-names>A</given-names>
</name>
<name>
<surname>Simonsen</surname> <given-names>A</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Ubukata</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Voice patterns as markers of schizophrenia: building a cumulative generalizable approach via a cross-linguistic and meta-analysis based investigation</article-title>. <source>Schizophr Bull</source>. (<year>2023</year>) <volume>49</volume>:<page-range>S125&#x2013;41</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/schbul/sbad046</pub-id>, PMID: <pub-id pub-id-type="pmid">36946527</pub-id></citation></ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alpert</surname> <given-names>M</given-names>
</name>
<name>
<surname>Rosenberg</surname> <given-names>SD</given-names>
</name>
<name>
<surname>Pouget</surname> <given-names>ER</given-names>
</name>
<name>
<surname>Shaw</surname> <given-names>RJ</given-names>
</name>
</person-group>. <article-title>Prosody and lexical accuracy in flat affect schizophrenia</article-title>. <source>Psychiatry Res</source>. (<year>2000</year>) <volume>97</volume>:<page-range>107&#x2013;18</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0165-1781(00)00211-0</pub-id>, PMID: <pub-id pub-id-type="pmid">11166083</pub-id></citation></ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hecker</surname> <given-names>P</given-names>
</name>
<name>
<surname>Steckhan</surname> <given-names>N</given-names>
</name>
<name>
<surname>Eyben</surname> <given-names>F</given-names>
</name>
<name>
<surname>Schuller</surname> <given-names>BW</given-names>
</name>
<name>
<surname>Arnrich</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Voice analysis for neurological disorder recognition&#x2013;a systematic review and perspective on emerging trends</article-title>. <source>Front Digital Health</source>. (<year>2022</year>) <volume>4</volume>:<elocation-id>842301</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fdgth.2022.842301</pub-id>, PMID: <pub-id pub-id-type="pmid">35899034</pub-id></citation></ref>
<ref id="B32">
<label>32</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cummins</surname> <given-names>N</given-names>
</name>
<name>
<surname>Matcham</surname> <given-names>F</given-names>
</name>
<name>
<surname>Klapper</surname> <given-names>J</given-names>
</name>
<name>
<surname>Schuller</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Chapter 10 - artificial intelligence to aid the detection of mood disorders</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Barh</surname> <given-names>D</given-names>
</name>
</person-group>, editor. <source>Artificial Intelligence in Precision Health</source>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>Academic Press</publisher-name> (<year>2020</year>). p. <page-range>231&#x2013;55</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/B978-0-12-817133-2.00010-0</pub-id>
</citation></ref>
<ref id="B33">
<label>33</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tzirakis</surname> <given-names>P</given-names>
</name>
<name>
<surname>Nguyen</surname> <given-names>A</given-names>
</name>
<name>
<surname>Zafeiriou</surname> <given-names>S</given-names>
</name>
<name>
<surname>Schuller</surname> <given-names>BW</given-names>
</name>
</person-group>. <source>ICASSP 2021&#x2013;2021 IEEE International Conference on Acoustics, Speech and Signal Processing</source>, <publisher-loc>Toronto, ON, Canada</publisher-loc> (<year>2021</year>). p. <page-range>6279&#x2013;83</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICASSP39728.2021.9414903</pub-id>
</citation></ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Poria</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cambria</surname> <given-names>E</given-names>
</name>
<name>
<surname>Bajpai</surname> <given-names>R</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>A review of affective computing: From unimodal analysis to multimodal fusion</article-title>. <source>Inf Fusion</source>. (<year>2017</year>) <volume>37</volume>:<fpage>98</fpage>&#x2013;<lpage>125</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inffus.2017.02.003</pub-id>
</citation></ref>
<ref id="B35">
<label>35</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zadeh</surname> <given-names>A</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>PP</given-names>
</name>
<name>
<surname>Poria</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cambria</surname> <given-names>E</given-names>
</name>
<name>
<surname>Morency</surname> <given-names>L-P</given-names>
</name>
</person-group>. (<year>2018</year>). <article-title>Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph</article-title>, in: <conf-name>Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics</conf-name> (<publisher-name>Association for Computational Linguistics</publisher-name>). pp. <page-range>2236&#x2013;46</page-range>.</citation></ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lian</surname> <given-names>H</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zong</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>A survey of deep learning-based multimodal emotion recognition: Speech, text, and face</article-title>. <source>Entropy</source>. (<year>2023</year>) <volume>25</volume>:<fpage>1440</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/e25101440</pub-id>, PMID: <pub-id pub-id-type="pmid">37895561</pub-id></citation></ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kossaifi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tzimiropoulos</surname> <given-names>G</given-names>
</name>
<name>
<surname>Todorovic</surname> <given-names>S</given-names>
</name>
<name>
<surname>Pantic</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Sewa db: A rich database for audio-visual emotion and sentiment research in the wild</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2019</year>) <volume>41</volume>:<page-range>615&#x2013;25</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2019.2944808</pub-id>, PMID: <pub-id pub-id-type="pmid">31581074</pub-id></citation></ref>
<ref id="B38">
<label>38</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jain</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Hugging face</article-title>. In: <source>Introduction to transformers for NLP: With the hugging face library and models to solve problems</source>. <publisher-name>Apress</publisher-name>, <publisher-loc>Berkeley, CA</publisher-loc> (<year>2022</year>). p. <fpage>51</fpage>&#x2013;<lpage>67</lpage>.</citation></ref>
<ref id="B39">
<label>39</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Eyben</surname> <given-names>F</given-names>
</name>
<name>
<surname>Weninger</surname> <given-names>F</given-names>
</name>
<name>
<surname>Gross</surname> <given-names>F</given-names>
</name>
<name>
<surname>Schuller</surname> <given-names>B</given-names>
</name>
</person-group>. (<year>2013</year>). <article-title>Recent developments in opensmile, the munich open-source multimedia feature extractor</article-title>, in: <conf-name>Proceedings of the 21st ACM international conference on Multimedia (ACM)</conf-name>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>. pp. <page-range>835&#x2013;8</page-range>.</citation></ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kanwal</surname> <given-names>S</given-names>
</name>
<name>
<surname>Asghar</surname> <given-names>S</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>A</given-names>
</name>
<name>
<surname>Rafique</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Identifying the evidence of speech emotional dialects using artificial intelligence: A cross-cultural study</article-title>. <source>PloS One</source>. (<year>2022</year>) <volume>17</volume>:<elocation-id>e0265199</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0265199</pub-id>, PMID: <pub-id pub-id-type="pmid">35298501</pub-id></citation></ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bitouk</surname> <given-names>D</given-names>
</name>
<name>
<surname>Verma</surname> <given-names>R</given-names>
</name>
<name>
<surname>Nenkova</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Class-level spectral features for emotion recognition</article-title>. <source>Speech Communication</source>. (<year>2010</year>) <volume>52</volume>:<page-range>613&#x2013;25</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.specom.2010.02.010</pub-id>, PMID: <pub-id pub-id-type="pmid">23794771</pub-id></citation></ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eyben</surname> <given-names>F</given-names>
</name>
<name>
<surname>Scherer</surname> <given-names>KR</given-names>
</name>
<name>
<surname>Schuller</surname> <given-names>BW</given-names>
</name>
<name>
<surname>Sundberg</surname> <given-names>J</given-names>
</name>
<name>
<surname>Andre</surname> <given-names>E</given-names>
</name>
<name>
<surname>Busso</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>The geneva minimalistic acoustic parameter set (gemaps) for voice research and affective computing</article-title>. <source>IEEE Trans Affect Computing</source>. (<year>2015</year>) <volume>7</volume>:<fpage>190</fpage>&#x2013;<lpage>202</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TAFFC.2015.2457417</pub-id>
</citation></ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chan</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>A method of prosodic assessment: Insights from a singing workshop</article-title>. <source>Cogent Educ</source>. (<year>2018</year>) <volume>5</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/2331186X.2018.1461047</pub-id>
</citation></ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kami&#x144;ska</surname> <given-names>D</given-names>
</name>
<name>
<surname>Sapi&#x144;ski</surname> <given-names>T</given-names>
</name>
<name>
<surname>Anbarjafari</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Efficiency of chosen speech descriptors in relation to emotion recognition</article-title>. <source>J Audio Speech Music Process</source>. (<year>2017</year>) <volume>2017</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13636-017-0100-x</pub-id>
</citation></ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rojo L&#xf3;pez</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Foulqui&#xe9;-Rubio</surname> <given-names>AI</given-names>
</name>
<name>
<surname>Espin Lopez</surname> <given-names>L</given-names>
</name>
<name>
<surname>Martinez Sanchez</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Analysis of speech rhythm and heart rate as indicators of stress on student interpreters</article-title>. <source>Perspectives</source>. (<year>2021</year>) <volume>29</volume>:<fpage>591</fpage>&#x2013;<lpage>607</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/0907676X.2021.1900305</pub-id>
</citation></ref>
<ref id="B46">
<label>46</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schuller</surname> <given-names>B</given-names>
</name>
<name>
<surname>Steidl</surname> <given-names>S</given-names>
</name>
<name>
<surname>Batliner</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>The interspeech 2009 emotion challenge</article-title>. In: <source>INTERSPEECH</source>. <publisher-loc>Brighton, UK</publisher-loc>: <publisher-name>ISCA</publisher-name> (<year>2009</year>). p. <page-range>312&#x2013;5</page-range>.</citation></ref>
<ref id="B47">
<label>47</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sweller</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Cognitive load theory</article-title>. In: <source>Psychology of Learning and Motivation</source>, vol. <volume>55</volume>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>Academic Press</publisher-name> (<year>2011</year>). p. <fpage>37</fpage>&#x2013;<lpage>76</lpage>.</citation></ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al Hanai</surname> <given-names>T</given-names>
</name>
<name>
<surname>Ghassemi</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Glass</surname> <given-names>JR</given-names>
</name>
</person-group>. <article-title>Detecting depression with audio/text sequence modeling of interviews</article-title>. <source>Proc Interspeech</source>. (<year>2018</year>), <page-range>1716&#x2013;20</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.21437/Interspeech.2018-2522</pub-id>
</citation></ref>
<ref id="B49">
<label>49</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Rozgi&#x107;</surname> <given-names>V</given-names>
</name>
<name>
<surname>Ananthakrishnan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Saleem</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>R</given-names>
</name>
<name>
<surname>Prasad</surname> <given-names>R</given-names>
</name>
</person-group>. (<year>2012</year>). <article-title>Ensemble of svm trees for multimodal emotion recognition</article-title>, in: <conf-name>In Proceedings of the 2012 Asia Pacific Signal and Information Processing Association Annual Summit and Conference (IEEE)</conf-name>, pp. <fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation></ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Fokou&#xe9;</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>A comparison of classifiers in performing speaker accent recognition using mfccs</article-title>. <source>arXiv preprint arXiv:1501.07866</source>. (<year>2015</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1501.07866</pub-id>
</citation></ref>
<ref id="B51">
<label>51</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kember</surname> <given-names>H</given-names>
</name>
<name>
<surname>Grohe</surname> <given-names>A-K</given-names>
</name>
<name>
<surname>Zahner</surname> <given-names>K</given-names>
</name>
<name>
<surname>Braun</surname> <given-names>B</given-names>
</name>
<name>
<surname>Weber</surname> <given-names>A</given-names>
</name>
<name>
<surname>Cutler</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Similar prosodic structure perceived differently in german and english</article-title>. In: <source>Interspeech</source>, vol. <volume>2017</volume>. (<year>2017</year>). p. <page-range>1388&#x2013;92</page-range>.</citation></ref>
<ref id="B52">
<label>52</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koolagudi</surname> <given-names>SG</given-names>
</name>
<name>
<surname>Rao</surname> <given-names>KS</given-names>
</name>
</person-group>. <article-title>Emotion recognition from speech: a review</article-title>. <source>Int J Speech Technol</source>. (<year>2012</year>) <volume>15</volume>:<fpage>99</fpage>&#x2013;<lpage>117</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10772-011-9125-1</pub-id>
</citation></ref>
<ref id="B53">
<label>53</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Riad</surname> <given-names>R</given-names>
</name>
<name>
<surname>Denais</surname> <given-names>M</given-names>
</name>
<name>
<surname>de Gennes</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lesage</surname> <given-names>A</given-names>
</name>
<name>
<surname>Oustric</surname> <given-names>V</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>Automated speech analysis for risk detection of depression, anxiety, insomnia, and fatigue: Algorithm development and validation study</article-title>. <source>J Med Internet Res</source>. (<year>2024</year>) <volume>26</volume>:<elocation-id>e58572</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.2196/58572</pub-id>, PMID: <pub-id pub-id-type="pmid">39324329</pub-id></citation></ref>
</ref-list>
</back>
</article>