<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article article-type="systematic-review" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychol.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Psychology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychol.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1664-1078</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyg.2025.1645860</article-id><article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading"><subject>Systematic Review</subject></subj-group>
</article-categories>
<title-group>
<article-title>Speech analysis and speech emotion recognition in mental disease: a scoping review</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes"><name><surname>Lombardo</surname> <given-names>Clara</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3015714"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
<contrib contrib-type="author"><name><surname>Esposito</surname> <given-names>Giulia</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3259017"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
<contrib contrib-type="author"><name><surname>Carbone</surname> <given-names>Silvia</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2720296"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
<contrib contrib-type="author"><name><surname>Serrano</surname> <given-names>Salvatore</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2426438"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
<contrib contrib-type="author"><name><surname>Mento</surname> <given-names>Carmela</given-names></name><xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/80680"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Department &#x201C;Scienze della Salute&#x201D;, University of Catanzaro</institution>, <city>Catanzaro</city>, <country country="it">Italy</country></aff>
<aff id="aff2"><label>2</label><institution>Department of Engineering, University of Messina</institution>, <city>Messina</city>, <country country="it">Italy</country></aff>
<aff id="aff3"><label>3</label><institution>Political and Legal Sciences Department, University of Messina</institution>, <city>Messina</city>, <country country="it">Italy</country></aff>
<aff id="aff4"><label>4</label><institution>Department of Biomedical and Dental Sciences and Morphofunctional Imaging, University of Messina</institution>, <city>Messina</city>, <country country="it">Italy</country></aff>
<author-notes><corresp id="c001"><label>&#x002A;</label>Correspondence: Clara Lombardo, <email xlink:href="mailto:clara.lombardo@unicz.it">clara.lombardo@unicz.it</email></corresp></author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-11-06">
<day>06</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1645860</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Lombardo, Esposito, Carbone, Serrano and Mento.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Lombardo, Esposito, Carbone, Serrano and Mento</copyright-holder>
<license><ali:license_ref start_date="2025-11-06">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec id="sec1">
<title>Background</title>
<p>Mental disorders have a significant impact on many areas of people&#x2019;s life, particularly on affective regulation; thus, there is a growing need to find disease-specific biomarkers to improve early diagnosis. Recently, machine learning technology using speech analysis proved to be a promising field that could aid mental health assessments. Furthermore, as prosodic expressions of emotions are altered in many psychiatric conditions, some studies successfully employed a speech emotion recognition model (SER) to identify mental diseases. The aim of this paper is to discuss the utilization of speech analysis in diagnosis of mental disorders, with a focus on studies using SER system to detect mental illness.</p>
</sec>
<sec id="sec2">
<title>Method</title>
<p>We searched PubMed, Scopus and Google Scholar for papers published from 2014 to 2024. We conducted a preliminary search, which revealed papers on the topic. Finally, 12 studies met the inclusion criteria and were included in the review.</p>
</sec>
<sec id="sec3">
<title>Results</title>
<p>Findings confirmed the efficacy of speech analysis in distinguishing between patients from healthy subjects; moreover, the examined studies underlined that some mental illnesses are associated with specific voice patterns. Furthermore, results from studies employing speech emotion recognition system to detect mental disorders showed that emotions can be successfully used as an intermediary step for mental diseases detection, particularly for mood disorders.</p>
</sec>
<sec id="sec4">
<title>Conclusion</title>
<p>These findings support the implementing of speech signals analysis in mental health assessment: it is an accessible and non-invasive method which can provide earlier diagnosis and a higher treatment personalization.</p>
</sec>
</abstract>
<kwd-group>
<kwd>speech analysis</kwd>
<kwd>acoustic features</kwd>
<kwd>speech emotion recognition</kwd>
<kwd>mental disorders</kwd>
<kwd>schizophrenia</kwd>
<kwd>depression</kwd>
</kwd-group><funding-group><funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the European Union - Next Generation EU under the Italian National Recovery and Resilience Plan (NRRP), Mission 4, Component 2, Investment 1.3, CUP C49J24000240004, partnership on &#x201C;Telecommunications of the Future&#x201D; (PE00000001 - program &#x201C;RESTART&#x201D;).</funding-statement></funding-group>
<counts>
<fig-count count="1"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="50"/>
<page-count count="10"/>
<word-count count="7630"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Emotion Science</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec5">
<label>1</label>
<title>Introduction</title>
<p>A psychiatric disorder is a mental or behavioral pattern that influences emotional regulation, behavior and cognition, causing a significant impairment in several areas of people&#x2019;s life, such as the functioning capacity at work and with their families (<xref ref-type="bibr" rid="ref25">Lalitha et al., 2021</xref>). In recent years, especially during the Covid-19 pandemic, there has been a significant increase in people affected by a mental disorder, with a consequent high impact on emotional life and affective regulation: about 970 million people in the world are currently suffering from a mental disorder and the number is expected to grow in the future (<xref ref-type="bibr" rid="ref5">Cansel et al., 2023</xref>). To now, there is still a lack of biomarkers and individualized treatment guidelines for mental illnesses (<xref ref-type="bibr" rid="ref7">Chen et al., 2022</xref>). In this regard, precision medicine is emerging in psychiatry as an innovative approach to improve the diagnosis and treatment of mental disorders, through a higher individualization of care and attention to the unique characteristics of each patient (<xref ref-type="bibr" rid="ref30">Manchia et al., 2020</xref>). Machine learning technology seems to be a promising field in mental health assessments: it may indeed be useful in screening of at-risk patients, improve the detection of disorder-specific features, allow to plan more efficient treatments and enable more real-time monitoring of psychiatric disorders (<xref ref-type="bibr" rid="ref29">Low et al., 2020</xref>; <xref ref-type="bibr" rid="ref41">Siena et al., 2020</xref>).</p>
<p>In particular, the language can be considered as a window into the mind (<xref ref-type="bibr" rid="ref24">Koops et al., 2023</xref>): people convey emotions, thoughts and motivations through speech (<xref ref-type="bibr" rid="ref51">Zhang et al., 2024</xref>). If the speech content is easily masked by people, features such as speed, energy and pitch variation in speech cannot be controlled. Therefore, vocal-acoustic cues allow to get an objective measurement of mental illness (<xref ref-type="bibr" rid="ref37">Patil and Wadhai, 2021</xref>). Many studies have demonstrated that acoustic parameters can be used as valid biomarkers for the early diagnosis of mental disorders (<xref ref-type="bibr" rid="ref34">Pan et al., 2019</xref>; <xref ref-type="bibr" rid="ref10">Cummins et al., 2015</xref>). The most common acoustic features analyzed are the spectral features, related to the energy or the spectral flatness, the prosodic features describing the speech intonation, rhythm and rate, the temporal characteristics (e.g., utterance duration, duration and number of pauses) and the cepstral features that are commonly used in speech recognition for their high performance in describing the variation of low frequencies of the signal (<xref ref-type="bibr" rid="ref20">Jiang et al., 2018</xref>; <xref ref-type="bibr" rid="ref45">Teixeira et al., 2023</xref>). These features can reflect emotional arousal and expressiveness. Specifically, the ones referred to prosody give information about speech emotional tone and dynamics of speech. For example, <xref ref-type="bibr" rid="ref18">Hashim et al. (2017)</xref> noticed that acoustic speech signals alterations characterizing spectrum and timing are useful to examine depressive symptoms levels and treatment effectiveness. Other studies have instead shown that depression was associated with changes in prosody, such as an overall speech rate (<xref ref-type="bibr" rid="ref1">Alghowinem et al., 2013</xref>; <xref ref-type="bibr" rid="ref49">Wang et al., 2021</xref>), and changes in speech spectrum, like the decrease in the sub-band energy variance (<xref ref-type="bibr" rid="ref10">Cummins et al., 2015</xref>). Furthermore, a recent meta-analysis on schizophrenic acoustic patterns showed that patients presented reduced speech rate and pitch variability (<xref ref-type="bibr" rid="ref35">Parola et al., 2018</xref>).</p>
<p>Negative emotions such as sadness, anger and fear are indicator of mental disorders (<xref ref-type="bibr" rid="ref26">Lalitha and Tripathi, 2016</xref>): for this reason, another promising approach to diagnosis of mental health conditions comes from Speech Emotion Recognition (SER), a system which provides an extraction of the speakers&#x2019; emotional states from their speech signals (<xref ref-type="bibr" rid="ref17">Hashem et al., 2023</xref>; <xref ref-type="bibr" rid="ref22">Kerkeni et al., 2019</xref>). It has been employed in detecting different mental illnesses, such as post-traumatic stress disorder (PTSD; <xref ref-type="bibr" rid="ref36">Pathan et al., 2023</xref>) and depression (<xref ref-type="bibr" rid="ref31">Mar and Pa, 2019</xref>). In particular, SER model utilization for depression prediction is supported by findings about the inhibition of prosodic emotional expression in depressive conditions, but also by experimental studies connecting positively SER and depression detection models (<xref ref-type="bibr" rid="ref42">Stasak et al., 2016</xref>). <xref ref-type="bibr" rid="ref16">Harati et al. (2018)</xref> carried out a computational speech analysis for classifying depression severity applying Deep Neural Network (DNN) model to audio recordings of patients with Major Depressive Disorder. Participants in this research are evaluated weekly for 8&#x202F;months, starting before Deep Brain Stimulation (DBS) and throughout the first 6&#x202F;months of DBS surgery; two clinical phases are therefore considered for the speech analysis: depressed and improved. This approach successfully classified the two phases of DBS treatment with an AUC of 0.80. Furthermore, <xref ref-type="bibr" rid="ref3">Bhavya et al. (2023)</xref> proposed a new computational methodology to detect different emotions and depression; specifically, they built a dataset for depression-related data using audio samples from the DAIC-WOZ depression dataset and the RAVDESS dataset, which includes a wide spectrum of emotions conveyed by speakers of both genders. This method proved to be useful to recognize depressive symptoms.</p>
<p>While several reviews have analyzed general aspects regarding the diagnostic use of speech as biomarker for the early diagnosis of mental disorders (<xref ref-type="bibr" rid="ref29">Low et al., 2020</xref>), few have particularly considered studies employing speech emotion recognition (SER) systems. This shows that there is a gap in synthesizing results specifically emerging from diagnostic studies employing SER. Therefore, the aim of this work is to supply an updated analysis of literature on acoustic features used as objective indicators for the diagnosis of mental disorders, with a focus on studies using speech emotion recognition system. This is in order to confirm the effectiveness of this approach in mental health assessments.</p>
</sec>
<sec sec-type="materials|methods" id="sec6">
<label>2</label>
<title>Materials and methods</title>
<p>This scoping review was conducted and reported in accordance with the PRISMA extension for Scoping Reviews (PRISMA-ScR) guidelines (<xref ref-type="bibr" rid="ref46">Tricco et al., 2018</xref>). The PRISMA-ScR flow diagram (<xref ref-type="fig" rid="fig1">Figure 1</xref>) illustrates the study selection process.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>PRISMA-ScR flow diagram.</p>
</caption>
<graphic xlink:href="fpsyg-16-1645860-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart illustrating a systematic review process. Identification stage shows records from PubMed (668), Google Scholar (15,000), and Scopus (186), with 854 duplicates removed. Screening stage left 14,962 records excluded. Eligibility stage reviewed 38 reports; 26 were excluded for various reasons. Twelve studies were included in the final synthesis.</alt-text>
</graphic>
</fig>
<sec id="sec7">
<label>2.1</label>
<title>Information sources and search strategy</title>
<p>We searched PubMed, Scopus and Google Scholar, for papers published from January 1, 2014 to November 1, 2024, with combinations of the following search terms: <italic>&#x201C;Speech analysis OR speech emotion recognition OR acoustic analysis OR acoustic features AND mental disorders AND schizophrenia AND depression AND bipolar disorder.&#x201D;</italic></p>
</sec>
<sec id="sec8">
<label>2.2</label>
<title>Data extraction</title>
<p>We conducted a preliminary search, which revealed papers on the topic. Articles were included in the review according to the following inclusion criteria: English language, only empirical studies (e.g., observational, non-randomized experimental and machine learning classification designs), studies involving clinical populations with mental disorders, studies that involved quantitative and/or qualitative assessments of the variables considered. Books, meta-analyses, and reviews were excluded; non-empirical studies and studies that did not involve quantitative and/or qualitative assessments of the variables were also excluded.</p>
</sec>
<sec id="sec9">
<label>2.3</label>
<title>Data synthesis</title>
<p>We found 15.854 articles. Of these, 854 were removed before screening since they were duplicates. At the first screening conducted by title and abstract, 14.962 studies were excluded. After the second screening conducted by full-text examination of 38 papers, 26 articles were excluded because they were reviews, meta-analysis, not specific, irrelevant for the topic, because full text was not available or because the trial did not present a control group. Finally, 12 studies met the inclusion criteria and were included in the review. Due to the high heterogeneity of the studies, a qualitative data analysis was conducted instead of a quantitative meta-analysis. The annexed table summarizes the selected articles (<xref ref-type="table" rid="tab1">Table 1</xref>), whereas the annexed flow diagram (<xref ref-type="fig" rid="fig1">Figure 1</xref>) summarizes the selection process.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Main results of included studies.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Authors</th>
<th align="left" valign="top">Aim</th>
<th align="left" valign="top">Sample</th>
<th align="left" valign="top">Materials and measures</th>
<th align="left" valign="top">Speech/ emotion analysis methods</th>
<th align="left" valign="top">Results</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref33">Mart&#x00ED;nez-S&#x00E1;nchez et al. (2015)</xref>
</td>
<td align="left" valign="top">Quantify the deficits in expressive prosody in schizophrenia and evaluate its discriminatory power between groups</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;80<break/>&#x2212;45 patients with schizophrenia (M&#x202F;=&#x202F;39.49, SD&#x202F;=&#x202F;10.89; 71.1% male)<break/>&#x2212;35 controls (M&#x202F;=&#x202F;35.34, SD&#x202F;=&#x202F;10.48; 62.9% male)</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Brief Psychiatric Rating Scale (BPRS)</p>
</list-item>
<list-item>
<p>Professional Fostex</p>
</list-item>
<list-item>
<p>FR-2LE recorder</p>
</list-item>
<list-item>
<p>Acoustic voice analysis 5.1.42 Praat program</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Acoustic voice analysis 5.1.42 Praat program for the extraction of different parameters (e.g., pitch, duration, temporal variations and pauses) related to expressive prosody.</p>
</list-item>
</list>
</td>
<td align="left" valign="top">Schizophrenic patients showed significantly more pauses (<italic>p</italic>&#x202F;&#x003C;&#x202F;0.001), less pitch variability in speech (<italic>p</italic>&#x202F;&#x003C;&#x202F;0.05), fewer variations in syllable timing (<italic>p</italic>&#x202F;&#x003C;&#x202F;0.001) and they were slower (<italic>p</italic>&#x202F;&#x003C;&#x202F;0.001) than control subjects. Signal processing algorithms applied to speech were shown an accuracy of 93.8% in distinguishing patients from healthy controls.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref40">Scherer et al., 2015</xref>
</td>
<td align="left" valign="top">Explore vowel space, a measure of frequency range, extracted from conversational Speech, and its relationship to self-reported symptoms of depression and post-traumatic stress disorder (PTSD)</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;253<break/><list list-type="bullet">
<list-item>
<p>Depression group&#x202F;=&#x202F;47 (33 male and 14 female)</p>
</list-item>
<list-item>
<p>PTSD group&#x202F;=&#x202F;88 (58 male and 30 female)</p>
</list-item>
<list-item>
<p>No depression group&#x202F;=&#x202F;205 (153 male and 52 female)</p>
</list-item>
<list-item>
<p>No PTSD group&#x202F;=&#x202F;165 (128 male and 37 female)</p>
</list-item>
</list></td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>PTSD Checklist-Civilian version (PCL-C)</p>
</list-item>
<list-item>
<p>Patient Health Questionnaire-Depression 9 (PHQ-9)</p>
</list-item>
<list-item>
<p>COVAREP toolbox for the processing of the speech signals</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>COVAREP toolbox for the processing of the speech signals (vowel space, formants, pitch, energy)</p>
</list-item>
</list>
</td>
<td align="left" valign="top">Results showed a significantly reduced vowel space in subjects that scored positively on the questionnaires of PTSD (PTSD <italic>M</italic>&#x202F;=&#x202F;0.51, non-PTSD <italic>M</italic>&#x202F;=&#x202F;0.56, <italic>t</italic>(251)&#x202F;=&#x202F;2.55, <italic>p</italic>&#x202F;=&#x202F;0.01 Hedges&#x2019; <italic>g</italic>&#x202F;=&#x202F;&#x2212;0.34) and depression (depressed <italic>M</italic>&#x202F;=&#x202F;0.49, non-depressed <italic>M</italic>&#x202F;=&#x202F;0.55, <italic>t</italic> (251)&#x202F;=&#x202F;2.69, <italic>p</italic>&#x202F;=&#x202F;0.008, Hedges&#x2019; <italic>g</italic>&#x202F;=&#x202F;&#x2212;0.43).</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref6">Chakraborty et al., 2018</xref>
</td>
<td align="left" valign="top">Employ low-level speech signals in the distinction of patients with schizophrenia from healthy individuals.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;78<break/><list list-type="bullet">
<list-item>
<p>52 patients with Schizophrenia</p>
</list-item>
</list>&#x2212;26 healthy controls</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Brief Assessment of Cognition (BAC)</p>
</list-item>
<list-item>
<p>Semi-structured clinical interview</p>
</list-item>
<list-item>
<p>NSA-16</p>
</list-item>
<list-item>
<p>OpenSMILE &#x2018;emobase&#x2019; to recognize emotion from acoustic signals.</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>OpenSMILE &#x201C;emobase&#x201D; to extract low-level prosodic features (e.g., intonation, energy, duration)</p>
</list-item>
</list>
</td>
<td align="left" valign="top">The objective openSMILE acoustic signals can be reliably used to distinguish between the patient and controls with an accuracy of 79&#x2013;86%.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref43">Stolar et al., 2018</xref>
</td>
<td align="left" valign="top">Detect depression with a clinical database of adolescents interacting with a parent.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;63<break/><list list-type="bullet">
<list-item>
<p>Depressed patients&#x202F;=&#x202F;29 (5 male and 24 female)</p>
</list-item>
<list-item>
<p>Healthy controls&#x202F;=&#x202F;34 (10 male and 24 female)</p>
</list-item>
</list></td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Voice activity detector (VAD) to extract voiced speech segments</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Voice activity detector (VAD) for the analysis of spectral parameters of speech (e.g., flux, centroids, formants and optimized spectral roll-off)</p>
</list-item>
</list>
</td>
<td align="left" valign="top">The proposed optimized feature set achieved an average depression detection accuracy of 82.2% for males and 70.5% for females. Among acoustic spectral features, the optimized spectral roll-off set is the most effective.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref44">Tahir et al., 2019</xref>
</td>
<td align="left" valign="top">Explore non-verbal speech signals as objective measures of negative symptoms of schizophrenia, studying the correlation with the subjective ratings of negative symptoms on a clinical scale.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;80<break/>&#x2212;54 patients with schizophrenia<break/>&#x2212;26 healthy controls</td>
<td align="left" valign="top"><list list-type="bullet">
<list-item>
<p>The Structured Clinical Interview for DSM-IV (SCID)</p>
</list-item>
</list>&#x2212;16-item Negative Symptom Assessment (NSA-16)</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Automatic extraction of non-verbal speech cues (e.g., pause duration, speech ratio, turn-taking and prosodic variability) through speech segmentation algorithms in Matlab.</p>
</list-item>
</list>
</td>
<td align="left" valign="top">The study allows to distinguish healthy and patients using non-verbal speech features (conversational and prosody related cues) with 81.3% accuracy.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref19">He et al., 2020</xref>
</td>
<td align="left" valign="top">Evaluate an automatic system for detecting the negative symptoms of patients with schizophrenia based on speech signal processing.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;56<break/>&#x2212;28 patients with schizophrenia (18 females and 10 males)<break/>&#x2212;28 healthy controls (18 females and 10 males)</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Psychotic Disorders Severity Scale</p>
</list-item>
<list-item>
<p>Reading three texts to express emotions and analyze the associated speech signals.</p>
</list-item>
<list-item>
<p>Decision tree for features classification</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Automatic acoustic signal processing system based on three features: SDVV (speech intensity), SSDL (spectral difference), QEVA (tone variation).</p>
</list-item>
</list>
</td>
<td align="left" valign="top">The most promising feature is the SDVV feature: it achieves an accuracy of more than 85% in the detection of schizophrenic patients&#x2019; speech in each emotional state. The combination of three acoustic features (SSDL, QEVA, SDVV) achieved a high level of accuracy (98.2%, with an AUC value of 98%) in discrimination of schizophrenic patients and controls.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref27">Lee et al., 2021</xref>
</td>
<td align="left" valign="top">Develop a voice-based screening test for depression<break/>using vocal acoustic features of elderly people, for males and females.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;204<break/><list list-type="bullet">
<list-item>
<p>Depressed patients&#x202F;=&#x202F;61</p>
</list-item>
<list-item>
<p>Healthy controls&#x202F;=&#x202F;143</p>
</list-item>
</list></td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Mini International Neuropsychiatric Interview (MINI-K)</p>
</list-item>
<list-item>
<p>Korean version of the Consortium to Establish a Registry for Alzheimer&#x2019;s Disease Assessment Packet Clinical Assessment Battery (CERAD-K-C)</p>
</list-item>
<list-item>
<p>Digit Span Test</p>
</list-item>
<list-item>
<p>Frontal Assessment Battery</p>
</list-item>
<list-item>
<p>Korean version of the geriatric depression scale (GDS-KR)</p>
</list-item>
<list-item>
<p>Mood-inducing sentences (MIS)</p>
</list-item>
<list-item>
<p>OpenSMILE v2.1.0 for speech analysis</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>OpenSMILE v2.1.0 to extract spectral, prosodic and energy features with AVEC 2013 and eGeMAPS sets</p>
</list-item>
</list>
</td>
<td align="left" valign="top">Acoustic features showing significant discriminatory performances are spectral and energy-related features for males (sensitivity 0.95, specificity 0.88, and accuracy 0.86) and prosody-related features for females (sensitivity 0.73, specificity 0.86, and accuracy 0.77).</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref37">Patil and Wadhai, 2021</xref>
</td>
<td align="left" valign="top">Use acoustic features extracted from the spontaneous speech samples of the volunteers to detect depression.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;129<break/><list list-type="bullet">
<list-item>
<p>Depressed patients&#x202F;=&#x202F;54</p>
</list-item>
<list-item>
<p>Healthy controls&#x202F;=&#x202F;75</p>
</list-item>
</list></td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Patient Health Questionnaire (PHQ-9)</p>
</list-item>
<list-item>
<p>Depression Inventory (BDI) scale</p>
</list-item>
<list-item>
<p>Praat 6.0 for speech analysis</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Praat v6.0 to extract parameters such as MFCC, pitch, jitter, shimmer, and energy</p>
</list-item>
<list-item>
<p>SVM, Random Forest, GMM classifiers for depression detection</p>
</list-item>
</list>
</td>
<td align="left" valign="top">Speech features like MFCC, pitch, jitter, shimmer and energy can be used as a reliable biomarker for depression detection.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref15">Hansen et al., 2022</xref>
</td>
<td align="left" valign="top">Explore a method to allow a clinical evaluation of depression and remission from acoustic speech</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;82<break/><list list-type="bullet">
<list-item>
<p>Healthy controls&#x202F;=&#x202F;42</p>
</list-item>
<list-item>
<p>Patients group&#x202F;=&#x202F;40 individuals with first-episode major depressive disorder (MDD)</p>
</list-item>
<list-item>
<p>Patients in remission&#x202F;=&#x202F;25</p>
</list-item>
</list></td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Hamilton Rating Scale for Depression, to evaluate depression severity and remission</p>
</list-item>
<list-item>
<p>Audio recordings of the Indiana Psychiatric Illness Interview</p>
</list-item>
<list-item>
<p>A gradient boosted decision tree model was trained to predict the probability of sounding happy or sad and combined in a Mixture of Experts (MoE) architecture for ensemble prediction</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>A gradient boosted decision tree model trained to predict the probability of sounding happy or sad and combined in a Mixture of Experts (MoE) architecture for ensemble prediction</p>
</list-item>
</list>
</td>
<td align="left" valign="top">Patients with depression have a probability of sounding sad (theta) of 0.70 (95% CI: 0.38, 0.90); patients in remission have a theta of 0.25 (95% CI: 0.07, 0.58); healthy controls at visit 1 have a theta of 0.23 (95% CI: 0.10, 0.47), and at visit 2 have a theta of 0.22 (95% CI: 0.07, 0.58). SER model allows to distinguish between depressed patients and healthy controls, achieving an AUC of 0.71.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref39">Rejaibi et al., 2022</xref>
</td>
<td align="left" valign="top">Present a deep Recurrent Neural Network-based framework to detect depression and to predict its severity level from speech.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;189<break/><list list-type="bullet">
<list-item>
<p>Depressed patients&#x202F;=&#x202F;56 (25 male and 31 female)</p>
</list-item>
<list-item>
<p>Healthy controls&#x202F;=&#x202F;133 (77 male and 56 female)</p>
</list-item>
</list></td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Patient Health Questionnaire of eight questions (PHQ-8)</p>
</list-item>
<list-item>
<p>DAIC-WOZ depression dataset</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>DAIC-WOZ depression dataset (extraction of low-level and high-level MFCC features from clinical audio)</p>
</list-item>
<list-item>
<p>RNN model for depression detection and PHQ-8 score prediction</p>
</list-item>
</list>
</td>
<td align="left" valign="top">The proposed approach obtained an accuracy of 76.27% in detecting depression. MFCC based high-level features give relevant information about depression.</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref48">Wanderley Espinola et al., 2022</xref>
</td>
<td align="left" valign="top">Present a methodology to support the diagnosis of schizophrenia, major depressive disorder, bipolar disorder, and generalized anxiety disorder using vocal acoustic analysis and machine learning.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;78<break/><list list-type="bullet">
<list-item>
<p>Depression group: 28 (17 males)</p>
</list-item>
<list-item>
<p>Schizophrenia group: 21 (12 males)</p>
</list-item>
<list-item>
<p>Bipolar Disorder group: 14</p>
</list-item>
<list-item>
<p>Generalized anxiety Disorder group: 4</p>
</list-item>
<list-item>
<p>Control group: 12 (7 males)</p>
</list-item>
</list></td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>Depression: HAM-D</p>
</list-item>
<list-item>
<p>Schizophrenia: BPRS</p>
</list-item>
<list-item>
<p>Bipolar disorder: YRMS</p>
</list-item>
<list-item>
<p>GAD: GAD-7</p>
</list-item>
<list-item>
<p>Control group: SRQ-20</p>
</list-item>
<list-item>
<p>Acquisition of voice samples: Tascam&#x2122; 16-bit linear PCM recorder</p>
</list-item>
<list-item>
<p>Audio editing: Audacity&#x2122; audio software</p>
</list-item>
<list-item>
<p>Feature extraction: GNU Octave&#x2122;</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>GNU Octave for the extraction of acoustic parameters such as pitch, intensity and formant bandwidths</p>
</list-item>
<list-item>
<p>Random Forest (300 trees) for the identification of four mental disorders</p>
</list-item>
</list>
</td>
<td align="left" valign="top">Forests with 300 trees attained the greatest discrimination performance (accuracy of 75.27% and kappa index of 0.6908).<break/><list list-type="bullet">
<list-item>
<p>Specifically, depression group got 0.713 of sensitivity, 0.925 of specificity, and 0.940 for area under ROC curve.</p>
</list-item>
<list-item>
<p>Schizophrenic group: 0.700 of sensitivity, 0.913 of specificity, and 0.929 for area under ROC curve.</p>
</list-item>
<list-item>
<p>Bipolar disorder: 0.830 for sensitivity, 0.952 for specificity, and 0.966 for area under ROC curve</p>
</list-item>
<list-item>
<p>Generalized anxiety disorder: 0.920 for sensitivity, 0.943 for specificity, and 0.985 for area under ROC curve.</p>
</list-item>
<list-item>
<p>Control group: 0.713 of sensitivity, 0.925 of specificity, and 0.940 for area under ROC curve.</p>
</list-item>
</list></td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref11">De Boer et al., 2023</xref>
</td>
<td align="left" valign="top">Estabilish the diagnostic potential of specific speech parameters in a sample of patients with a schizophrenia-spectrum disorder and analyze the ability of acoustic analyses in differentiating between patients who experience predominantly positive versus negative psychotic symptoms.</td>
<td align="left" valign="top"><italic>N</italic>&#x202F;=&#x202F;284<break/>&#x2212;142 patients with a schizophrenia-spectrum disorder<break/>&#x2212;142 matched controls</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>PANSS</p>
</list-item>
<list-item>
<p>Semi-structured interviews</p>
</list-item>
<list-item>
<p>OpenSMILE for speech analysis</p>
</list-item>
</list>
</td>
<td align="left" valign="top">
<list list-type="bullet">
<list-item>
<p>OpenSMILE for the acoustic feature extraction (frequency, energy, spectral and temporal) using the extended Geneva Minimalistic Acoustic Parameter Set (eGeMAPS)</p>
</list-item>
<list-item>
<p>ML classification for schizophrenia</p>
</list-item>
</list>
</td>
<td align="left" valign="top">The machine-learning achieved an accuracy of 86.2% (AUC of 0.92) in identifying patients with a schizophrenia-spectrum disorder and healthy controls. Moreover, it allowed to classify patients with predominantly positive or negative symptoms with an accuracy of 74.2% (AUC&#x2013;ROC of 0.76). 10 acoustic parameters had the highest importance scores in the final model (<italic>p</italic> value &#x003C;0.001).</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="results" id="sec10">
<label>3</label>
<title>Results</title>
<p>Twelve empirical studies were found through literature search (see <xref ref-type="table" rid="tab1">Table 1</xref>), including case&#x2013;control, cross-sectional, longitudinal and ML-based classification designs. The majority of studies were about a specific psychiatric disorder (5 were conducted on schizophrenic patients and 5 on depressed ones); one of them focused on two disorders (depression and PTSD) and another one considered four diagnostic categories (major depressive disorder, bipolar disorder, schizophrenia or generalized anxiety disorder). Generally, they used vocal acoustic analysis and machine learning to analyze several categories of acoustic features; four studies instead employed speech emotion recognition model.</p>
<sec id="sec11">
<label>3.1</label>
<title>Schizophrenia</title>
<p>A study of <xref ref-type="bibr" rid="ref33">Mart&#x00ED;nez-S&#x00E1;nchez et al. (2015)</xref>, conducted on 45 patients with schizophrenia and 35 healthy controls, showed that schizophrenic patients generally present less pitch variability in speech, make more pauses and show a significantly lower voice intensity than controls: they therefore exhibited a prosodic and melodically flatter speech. <xref ref-type="bibr" rid="ref11">De Boer et al. (2023)</xref> found a top 10 of acoustic parameters that allowed to distinguish 142 patients with a schizophrenia-spectrum disorder from 142 matched controls. In particular, patients were classified using temporal features, such as a fragmented speech and longer pauses, and spectral ones, such as a reduced mean spectral slope and spectral flux variation, which, respectively, indicate a more tensed and monotonous voice in the patients. Moreover, some of these speech parameters can be useful to identify subjects with predominant positive or negative symptoms in schizophrenia-spectrum disorders: subjects with positive symptoms presented less variation in jitter (indicating rough voice), reduced variation in vowel frequency and a smaller F1 and F2 formant bandwidth (indicating breathiness). Other studies employed acoustic parameters to identify negative symptoms of schizophrenia; for example, <xref ref-type="bibr" rid="ref44">Tahir et al. (2019)</xref> analyzed non-verbal speech signals (e.g., prosodic and conversational cues) as objective measures of negative symptoms of schizophrenia, obtaining significant correlations between these features and specific indicators of the 16-item Negative Symptom Assessment (NSA-16)&#x2014;a semi-structured interview used to measure the severity of negative symptoms. A promising automatic procedure to detect the affective flattening, which is a typical negative symptoms of schizophrenia, was proposed in a study of <xref ref-type="bibr" rid="ref19">He et al. (2020)</xref>, conducted on 56 subjects (28 patients and 28 healthy controls); it was based on three speech characteristics: the symmetric spectral difference level (SSDL), useful to study spectral differences related to emotional richness, the quantization error and vector angle (QEVA), which reflect the variations in tone, and the standard dynamic volume value (SDVV), representing the modulation of speech intensity. The most promising feature is the SDVV feature: it achieved an accuracy of more than 85% in recognizing schizophrenic patients&#x2019; speech in each emotional state (especially the &#x201C;afraid&#x201D; and &#x201C;happy&#x201D; states); however, the combination of these acoustic features achieved a higher level of accuracy (98.2%) in detecting schizophrenia (<xref ref-type="bibr" rid="ref19">He et al., 2020</xref>).</p>
</sec>
<sec id="sec12">
<label>3.2</label>
<title>Depression</title>
<p><xref ref-type="bibr" rid="ref37">Patil and Wadhai (2021)</xref> extracted several acoustic features from the spontaneous speech of 129 participants (54 depressed patients and 75 controls) using different classifiers; results demonstrated that some of these parameters, such as MFCC, pitch, jitter (a measure of frequency instability), shimmer (related to amplitude variation in voice) and energy, can be successfully used as reliable biomarkers for depression assessment. <xref ref-type="bibr" rid="ref39">Rejaibi et al. (2022)</xref> proposed an MFCC-based Recurrent Neural Network to detect depression and to assess its severity level from speech: low-level and high-level audio features are extracted from 189 audio recordings (56 patients with depression and 133 healthy controls) to predict the 24 scores of the Patient Health Questionnaire (PHQ-8). They showed that MFCC-based high-level features provided significant information related to depression. <xref ref-type="bibr" rid="ref43">Stolar et al. (2018)</xref> analyzed adolescent depression detection from a clinical database of 63 adolescents (29 depressed patients and 34 controls) interacting with a parent. Many spectral parameters were investigated (i.e., flux, centroid, formants and power spectral density) to identify depression; however, the optimized spectral roll-off set, which represents the frequency-energy relationship, proved to be the most effective compared to other spectral features to detect depression. In a study of <xref ref-type="bibr" rid="ref27">Lee et al. (2021)</xref> conducted on 61 elderly Koreans with major depressive disorder (MDD), a gender difference was found in acoustic features related to depression: acoustic characteristics with considerable discriminatory performances concerned prosody in females and speech spectrum and energy in males; in particular, males with MDD presented lower loudness compared to controls.</p>
</sec>
<sec id="sec13">
<label>3.3</label>
<title>Other mental disorders</title>
<p>Two studies considered more diagnostic disorders. Specifically, <xref ref-type="bibr" rid="ref40">Scherer et al. (2015)</xref> examinated an automatic unsupervised machine learning based approach to detect vowel space, a measure of frequency range related to vowel articulation, extracted from the conversational speech of 256 individuals (47 depressed patients, 88 patients with PTSD, 205 no-depressed and 165 no-PTSD patients). Findings showed that subjects with depression and PTSD presented a significantly reduced vowel space. <xref ref-type="bibr" rid="ref48">Wanderley Espinola et al. (2022)</xref> instead proposed a methodology to support the diagnosis of schizophrenia, major depressive disorder (MDD), bipolar disorder (BD), and generalized anxiety disorder using vocal acoustic analysis and machine learning. They found that some vocal characteristics are unique for a specific group whereas others are shared by different groups. For instance, an increased pitch variability and increased intensity/volume are typical in bipolar disorder; reduced pitch range occurs both in depression and schizophrenia.</p>
</sec>
<sec id="sec14">
<label>3.4</label>
<title>Speech emotion recognition</title>
<p>Four of the analyzed studies used an emotion recognition model to evaluate mood disorders or negative symptoms of schizophrenic patients, demonstrating that it is a promising method in diagnosis of mental diseases. A commonly used open-source feature extraction toolkit for speech emotion recognition is OpenSMILE. For instance, <xref ref-type="bibr" rid="ref11">De Boer et al. (2023)</xref> extracted four types of acoustic parameters with OpenSMILE, employing the extended Geneva Acoustic Minimalistic Parameter Set (eGeMAPS; <xref ref-type="bibr" rid="ref12">Eyben et al., 2015</xref>): energy/amplitude, frequency, temporal and spectral features; the trained classifier achieved an accuracy of 86.2% (AUC of 0.92) in distinguishing schizophrenia-spectrum patients from controls. In <xref ref-type="bibr" rid="ref27">Lee et al. (2021)</xref> speech data were analyzed using two emotion recognition sets, the Audio-Visual Emotion Challenge 2013 (AVEC 2013) audio baseline feature set (<xref ref-type="bibr" rid="ref47">Valstar et al., 2013</xref>) and the extended Geneva Minimalistic Acoustic Parameter Set (eGeMAPS). <xref ref-type="bibr" rid="ref6">Chakraborty et al. (2018)</xref> used low-level acoustic prosodic features to distinguish between 52 individuals with schizophrenia and 26 healthy subjects and accurately detect the presence and severity of negative symptoms; furthermore, results showed that the subjective valuations of NSA-16 (16 items-Negative Symptom Assessment) could be precisely predicted from the objective acoustic features extracted with OpenSMILE &#x201C;emobase&#x201D; set. Finally, <xref ref-type="bibr" rid="ref15">Hansen et al. (2022)</xref> employed a Mixture-of-Experts machine learning model to recognize two emotional states (happy and sad) using three available emotional speech datasets in German and English. They demonstrated how this speech emotion recognition model allows to detect modifications in depressed patients&#x2019; speech before and after remission; specifically, depressed patients had a higher probability of sounding sad than controls, whereas the voice of patients in remission was more happy sounding compared to the period of disease.</p>
</sec>
<sec id="sec15">
<label>3.5</label>
<title>Risk of bias</title>
<p>No formal assessment of bias risk was conducted, as this is not required for scoping reviews according to the PRISMA-ScR guidelines. However, potential biases were considered narratively. Two reviewers independently analyzed the studies, discussing any discrepancies until consensus was reached. While acknowledging that the inclusion of only recent articles in English may have introduced selection bias, the results were interpreted with caution and methodological transparency.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec16">
<label>4</label>
<title>Discussion</title>
<p>A total of 12 studies that evaluate acoustic parameters from speech to detect clinical disorders were reviewed; all of them confirm results of previous works, showing that acoustic features can be valid biomarkers of mental disorders (<xref ref-type="bibr" rid="ref44">Tahir et al., 2019</xref>; <xref ref-type="bibr" rid="ref43">Stolar et al., 2018</xref>). Beyond supporting the validity of speech signals analysis in detecting a mental disorder, these studies also highlighted that some mental illnesses are associated with specific voice patterns and specific changes in speech prosody or spectrum.</p>
<p>For instance, schizophrenic patients present prosodic and melodically flatter speech, decreased spectral slope (indicating more tension in the voice) and show a significantly lower voice intensity than healthy controls (<xref ref-type="bibr" rid="ref33">Mart&#x00ED;nez-S&#x00E1;nchez et al., 2015</xref>); they also show fragmented speech and make more pauses than control group. Temporal parameters proved to be very important in identifying patients and controls: for instance, reduced speech rate can be related to slower processing speed or slower articulation (<xref ref-type="bibr" rid="ref8">&#x00C7;okal et al., 2019</xref>). Furthermore, some speech parameters can be used to distinguish between patients with positive and negative symptoms in schizophrenia-spectrum disorders. Subjects with positive symptoms generally present less variation in jitter, differences in vowel quality and a lower F1 formant frequency (<xref ref-type="bibr" rid="ref11">De Boer et al., 2023</xref>); characteristics as low-level acoustic prosodic features and three speech parameters, related, respectively, to spectral signals (SSDL), variations in speech tone (QEVA) and intensity (SDVV), allow instead to identify negative symptoms. The anomalies of affective prosody are indeed directly related to the blunting of emotional affect (<xref ref-type="bibr" rid="ref6">Chakraborty et al., 2018</xref>); moreover, since SDVV feature is related to speech emotional fluctuation, it can help in detecting monotonous speech, which is typical of schizophrenic patients with affective flattening (<xref ref-type="bibr" rid="ref19">He et al., 2020</xref>) These results are in line with previous research showing that schizophrenia is associated with a general lower vocal expressivity, reduced variations in vocal pitch and energy, and lower speed (<xref ref-type="bibr" rid="ref38">Rapcan et al., 2010</xref>; <xref ref-type="bibr" rid="ref9">Compton et al., 2018</xref>).</p>
<p>Also depression is related to different changes in acoustic parameters, such as MFCC, pitch, jitter, shimmer and energy (<xref ref-type="bibr" rid="ref39">Rejaibi et al., 2022</xref>): in particular, depressed patients&#x2019; speech is characterized by higher range of jitter and lower shimmer compared to healthy controls (<xref ref-type="bibr" rid="ref37">Patil and Wadhai, 2021</xref>); furthermore, lower voice energy can be considered a clinical manifestation of depression (<xref ref-type="bibr" rid="ref32">Marmor et al., 2016</xref>). A study showed that acoustic features discriminating depressed patients from control group were different in males and females: spectrum and energy-related features were specific in males and prosody-related features (e.g., F0) in females; since F0 is influenced by hormonal changes occurring in females, and since estrogens are associated with a higher incidence of depression in females, it can be assumed that this feature represents the physiology of depressive disorder in females (<xref ref-type="bibr" rid="ref27">Lee et al., 2021</xref>).</p>
<p>An interesting parameter associated with depression and PTSD is the significantly reduced vowel space, a measure of frequency range related to vowel articulation; this characteristic is probably due to the typical psychomotor retardation influencing motor control, that is a common symptom of Parkinson&#x2019;s disease too (<xref ref-type="bibr" rid="ref40">Scherer et al., 2015</xref>). A study has instead confirmed that depression shares some acoustic characteristics with schizophrenia, such as reduced pitch range (<xref ref-type="bibr" rid="ref48">Wanderley Espinola et al., 2022</xref>). These results confirm the previous literature that showed how depressed patients&#x2019; speech generally presents monotonous loudness and pitch (<xref ref-type="bibr" rid="ref13">France et al., 2000</xref>) and lower articulation rate (<xref ref-type="bibr" rid="ref4">Cannizzaro et al., 2004</xref>; <xref ref-type="bibr" rid="ref2">Alpert et al., 2001</xref>).</p>
<p>However, unlike former reviews that primarily focused on the vocal parameters of mental disorders such as schizophrenia or depression, this work broadens the scope of research by focusing specifically on studies that have employed SER systems. This represents a very innovative approach, little explored in previous reviews (<xref ref-type="bibr" rid="ref21">Jordan et al., 2025</xref>). A part of the examined studies is indeed focused on detecting mental disorders through speech emotion recognition system: the majority of them employed OpenSMILE toolkit, which proved to be useful to extract several emotion-related acoustic parameters from participants&#x2019; speech, such as temporal, spectral and prosodic ones (<xref ref-type="bibr" rid="ref11">De Boer et al., 2023</xref>; <xref ref-type="bibr" rid="ref27">Lee et al., 2021</xref>). These studies consistently underline how emotional prosody can be considered an intermediary between acoustic features and specific psychiatric symptoms, especially in mood and schizophrenia spectrum disorders. For example, depressed patients generally show a more sad sounding compared to healthy controls: this reflects the anhedonia and neurovegetative symptoms which are typical of melancholic subtype of depression (<xref ref-type="bibr" rid="ref15">Hansen et al., 2022</xref>); these findings are consistent with results of a study from <xref ref-type="bibr" rid="ref23">Khorram et al. (2018)</xref> on 12 patients with bipolar disorder, which showed that manic states are related to more positive and activated emotions compared to depressed states. Moreover, low-level acoustic signals proved to be a significant mark of affective prosody dysfunction in schizophrenia (<xref ref-type="bibr" rid="ref6">Chakraborty et al., 2018</xref>).</p>
<p>Although a quantitative synthesis was not conducted, the qualitative analysis of the data still allows us to identify differences and similarities between the studies examined. However, this work presents some limitations. First, the analyzed studies show a great heterogeneity as regards methods: in some of them participants are recorded during an interview with a psychologist (<xref ref-type="bibr" rid="ref44">Tahir et al., 2019</xref>; <xref ref-type="bibr" rid="ref6">Chakraborty et al., 2018</xref>); others employed text reading which elicited emotions (<xref ref-type="bibr" rid="ref19">He et al., 2020</xref>; <xref ref-type="bibr" rid="ref33">Mart&#x00ED;nez-S&#x00E1;nchez et al., 2015</xref>) or mood-inducing sentences (MIS; <xref ref-type="bibr" rid="ref27">Lee et al., 2021</xref>). Moreover several features extraction methods, such as OpenSMILE (<xref ref-type="bibr" rid="ref6">Chakraborty et al., 2018</xref>; <xref ref-type="bibr" rid="ref27">Lee et al., 2021</xref>; <xref ref-type="bibr" rid="ref11">De Boer et al., 2023</xref>) or other open software for speech analysis (<xref ref-type="bibr" rid="ref37">Patil and Wadhai, 2021</xref>; <xref ref-type="bibr" rid="ref48">Wanderley Espinola et al., 2022</xref>; <xref ref-type="bibr" rid="ref40">Scherer et al., 2015</xref>) were employed, and different classifiers, like a simple decision tree (<xref ref-type="bibr" rid="ref19">He et al., 2020</xref>) or Random Forest algorithm, Support Vector Machine (SVM), Gaussian Mixture Model (GMM; <xref ref-type="bibr" rid="ref37">Patil and Wadhai, 2021</xref>; <xref ref-type="bibr" rid="ref44">Tahir et al., 2019</xref>; <xref ref-type="bibr" rid="ref6">Chakraborty et al., 2018</xref>) have been applied. Finally, of all studies, only one was longitudinal, and only two considered more diagnostic categories.</p>
<p>New developments in active learning offer a promising strategy to address the two most important challenges raised in this work related to speech emotion recognition (SER) &#x2013; limited labeled data availability and class imbalance. Recent studies have proposed several effective SER frameworks that not only reduce labeling costs but also improve emotion recognition accuracy. <xref ref-type="bibr" rid="ref28">Li et al. (2023)</xref>, for example, introduced the AFTER framework that combines iterative sample selection with adaptive fine-tuning; furthermore, <xref ref-type="bibr" rid="ref14">Han et al. (2013)</xref> showed that active learning can also be effectively used in dimensional emotion recognition, demonstrating that selecting more uncertain samples allows maintaining performance similar to that of fully supervised models while using 12% less labeled data in offline (pool-based) systems and 6&#x2013;11% less labeled data in online (stream-based) systems, respectively.</p>
</sec>
<sec sec-type="conclusions" id="sec17">
<label>5</label>
<title>Conclusion</title>
<p>Findings support the utilization of speech analysis to detect several psychiatric disorders: it is accessible, non-invasive and can provide earlier diagnosis along with higher treatment personalization. All the studies analyzed indeed confirm that acoustic features can be used as valid biomarkers of mental disorders. Furthermore, some of them underlined that some mental diseases are associated with specific alterations in speech prosody or spectrum: specifically, depressive speech is characterized by monotonous loudness and pitch, lower speech rate and a sadder sounding. MFCC based high-level features offer significant information about depression. Some of these features occur in schizophrenia too, in particular in schizophrenic patients with affective flattening: they generally show prosodic and melodically flatter speech, a lower voice intensity, fragmented speech, and make more pauses compared to healthy controls. Although many of the findings confirm those of previous studies on the acoustic features of mental disorders, the innovative aspect of this review is to offer an updated analysis not only of the diagnostic value of vocal parameters in general, but also of SER systems. The integration of these two lines of research represents a promising perspective for early diagnosis in psychiatry through various speech biomarkers. Future studies should work on larger samples and evaluate clinical implications of these procedures in longitudinal studies; moreover, trans-diagnostic studies could allow to better identify disorders-specific acoustic features, as well as improve generalization.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec18">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article, further requests for information may be addressed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="sec19">
<title>Author contributions</title>
<p>CL: Formal analysis, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. GE: Conceptualization, Data curation, Formal analysis, Methodology, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. SC: Supervision, Writing &#x2013; review &#x0026; editing. SS: Data curation, Supervision, Writing &#x2013; review &#x0026; editing. CM: Conceptualization, Data curation, Supervision, Writing &#x2013; review &#x0026; editing.</p>
</sec>

<sec sec-type="COI-statement" id="sec21">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec22">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec23">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec24">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyg.2025.1645860/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fpsyg.2025.1645860/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Alghowinem</surname> <given-names>S.</given-names></name> <name><surname>Goecke</surname> <given-names>R.</given-names></name> <name><surname>Wagner</surname> <given-names>M.</given-names></name> <name><surname>Epps</surname> <given-names>J.</given-names></name> <name><surname>Breakspear</surname> <given-names>M.</given-names></name> <name><surname>Parker</surname> <given-names>G.</given-names></name></person-group> (<year>2013</year>). &#x201C;Detecting depression: a comparison between spontaneous and read speech.&#x201D; In <italic>2013 IEEE International Conference on Acoustics, Speech and Signal Processing</italic> (pp. 7547&#x2013;7551). IEEE.</mixed-citation></ref>
<ref id="ref2"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alpert</surname> <given-names>M.</given-names></name> <name><surname>Pouget</surname> <given-names>E. R.</given-names></name> <name><surname>Silva</surname> <given-names>R. R.</given-names></name></person-group> (<year>2001</year>). <article-title>Reflections of depression in acoustic measures of the patient&#x2019;s speech</article-title>. <source>J. Affect. Disord.</source> <volume>66</volume>, <fpage>59</fpage>&#x2013;<lpage>69</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0165-0327(00)00335-9</pub-id>, PMID: <pub-id pub-id-type="pmid">11532533</pub-id></mixed-citation></ref>
<ref id="ref3"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Bhavya</surname> <given-names>S.</given-names></name> <name><surname>Nayak</surname> <given-names>D. S.</given-names></name> <name><surname>Dmello</surname> <given-names>R. C.</given-names></name> <name><surname>Nayak</surname> <given-names>A.</given-names></name> <name><surname>Bangera</surname> <given-names>S. S.</given-names></name></person-group> (<year>2023</year>), &#x201C;Machine learning applied to speech emotion analysis for depression recognition.&#x201D; In <italic>2023 international conference for advancement in technology (ICONAT)</italic> (pp. 1&#x2013;5). IEEE.</mixed-citation></ref>
<ref id="ref4"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cannizzaro</surname> <given-names>M.</given-names></name> <name><surname>Harel</surname> <given-names>B.</given-names></name> <name><surname>Reilly</surname> <given-names>N.</given-names></name> <name><surname>Chappell</surname> <given-names>P.</given-names></name> <name><surname>Snyder</surname> <given-names>P. J.</given-names></name></person-group> (<year>2004</year>). <article-title>Voice acoustical measurement of the severity of major depression</article-title>. <source>Brain Cogn.</source> <volume>56</volume>, <fpage>30</fpage>&#x2013;<lpage>35</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bandc.2004.05.003</pub-id>, PMID: <pub-id pub-id-type="pmid">15380873</pub-id></mixed-citation></ref>
<ref id="ref5"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cansel</surname> <given-names>N.</given-names></name> <name><surname>Alcin</surname> <given-names>&#x00D6;. F.</given-names></name> <name><surname>Y&#x0131;lmaz</surname> <given-names>&#x00D6;. F.</given-names></name> <name><surname>Ari</surname> <given-names>A.</given-names></name> <name><surname>Akan</surname> <given-names>M.</given-names></name> <name><surname>Ucuz</surname> <given-names>&#x0130;.</given-names></name></person-group> (<year>2023</year>). <article-title>A new artificial intelligence-based clinical decision support system for diagnosis of major psychiatric diseases based on voice analysis</article-title>. <source>Psychiatr. Danub.</source> <volume>35</volume>, <fpage>489</fpage>&#x2013;<lpage>499</lpage>. doi: <pub-id pub-id-type="doi">10.24869/psyd.2023.489</pub-id>, PMID: <pub-id pub-id-type="pmid">37992093</pub-id></mixed-citation></ref>
<ref id="ref6"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Chakraborty</surname> <given-names>D.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Tahir</surname> <given-names>Y.</given-names></name> <name><surname>Maszczyk</surname> <given-names>T.</given-names></name> <name><surname>Dauwels</surname> <given-names>J.</given-names></name></person-group>, (<year>2018</year>). &#x201C;Prediction of negative symptoms of schizophrenia from emotion related low-level speech signals.&#x201D; In <italic>2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</italic> (pp. 6024&#x2013;6028). IEEE.</mixed-citation></ref>
<ref id="ref7"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z. S.</given-names></name> <name><surname>Galatzer-Levy</surname> <given-names>I. R.</given-names></name> <name><surname>Bigio</surname> <given-names>B.</given-names></name> <name><surname>Nasca</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Modern views of machine learning for precision psychiatry</article-title>. <source>Patterns</source> <volume>3</volume>:<fpage>100602</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.patter.2022.100602</pub-id>, PMID: <pub-id pub-id-type="pmid">36419447</pub-id></mixed-citation></ref>
<ref id="ref8"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>&#x00C7;okal</surname> <given-names>D.</given-names></name> <name><surname>Zimmerer</surname> <given-names>V.</given-names></name> <name><surname>Turkington</surname> <given-names>D.</given-names></name> <name><surname>Ferrier</surname> <given-names>N.</given-names></name> <name><surname>Varley</surname> <given-names>R.</given-names></name> <name><surname>Watson</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Disturbo del ritmo del pensiero: modelli di pausa del linguaggio nella schizofrenia, con e senza disturbo formale del pensiero</article-title>. <source>PLoS One</source> <volume>14</volume>:<fpage>e0217404</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0217404</pub-id></mixed-citation></ref>
<ref id="ref9"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Compton</surname> <given-names>M. T.</given-names></name> <name><surname>Lunden</surname> <given-names>A.</given-names></name> <name><surname>Cleary</surname> <given-names>S. D.</given-names></name> <name><surname>Pauselli</surname> <given-names>L.</given-names></name> <name><surname>Alolayan</surname> <given-names>Y.</given-names></name> <name><surname>Halpern</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>The aprosody of schizophrenia: computationally derived acoustic phonetic underpinnings of monotone speech</article-title>. <source>Schizophr. Res.</source> <volume>197</volume>, <fpage>392</fpage>&#x2013;<lpage>399</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.schres.2018.01.007</pub-id>, PMID: <pub-id pub-id-type="pmid">29449060</pub-id></mixed-citation></ref>
<ref id="ref10"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cummins</surname> <given-names>N.</given-names></name> <name><surname>Scherer</surname> <given-names>S.</given-names></name> <name><surname>Krajewski</surname> <given-names>J.</given-names></name> <name><surname>Schnieder</surname> <given-names>S.</given-names></name> <name><surname>Epps</surname> <given-names>J.</given-names></name> <name><surname>Quatieri</surname> <given-names>T. F.</given-names></name></person-group> (<year>2015</year>). <article-title>A review of depression and suicide risk assessment using speech analysis</article-title>. <source>Speech Comm.</source> <volume>71</volume>, <fpage>10</fpage>&#x2013;<lpage>49</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.specom.2015.03.004</pub-id></mixed-citation></ref>
<ref id="ref11"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>De Boer</surname> <given-names>J. N.</given-names></name> <name><surname>Voppel</surname> <given-names>A. E.</given-names></name> <name><surname>Brederoo</surname> <given-names>S. G.</given-names></name> <name><surname>Schnack</surname> <given-names>H. G.</given-names></name> <name><surname>Truong</surname> <given-names>K. P.</given-names></name> <name><surname>Wijnen</surname> <given-names>F. N. K.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Acoustic speech markers for schizophrenia-spectrum disorders: a diagnostic and symptom-recognition tool</article-title>. <source>Psychol. Med.</source> <volume>53</volume>, <fpage>1302</fpage>&#x2013;<lpage>1312</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0033291721002804</pub-id>, PMID: <pub-id pub-id-type="pmid">34344490</pub-id></mixed-citation></ref>
<ref id="ref12"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eyben</surname> <given-names>F.</given-names></name> <name><surname>Scherer</surname> <given-names>K. R.</given-names></name> <name><surname>Schuller</surname> <given-names>B. W.</given-names></name> <name><surname>Sundberg</surname> <given-names>J.</given-names></name> <name><surname>Andr&#x00E9;</surname> <given-names>E.</given-names></name> <name><surname>Busso</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>The Geneva minimalistic acoustic parameter set (GeMAPS) for voice research and affective computing</article-title>. <source>IEEE Trans. Affect. Comput.</source> <volume>7</volume>, <fpage>190</fpage>&#x2013;<lpage>202</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TAFFC.2015.2457417</pub-id></mixed-citation></ref>
<ref id="ref13"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>France</surname> <given-names>D. J.</given-names></name> <name><surname>Shiavi</surname> <given-names>R. G.</given-names></name> <name><surname>Silverman</surname> <given-names>S.</given-names></name> <name><surname>Silverman</surname> <given-names>M.</given-names></name> <name><surname>Wilkes</surname> <given-names>M.</given-names></name></person-group> (<year>2000</year>). <article-title>Acoustical properties of speech as indicators of depression and suicidal risk</article-title>. <source>IEEE Trans. Biomed. Eng.</source> <volume>47</volume>, <fpage>829</fpage>&#x2013;<lpage>837</lpage>. doi: <pub-id pub-id-type="doi">10.1109/10.846676</pub-id>, PMID: <pub-id pub-id-type="pmid">10916253</pub-id></mixed-citation></ref>
<ref id="ref14"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Ruan</surname> <given-names>H.</given-names></name> <name><surname>Ma</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Schuller</surname> <given-names>B. W.</given-names></name></person-group> (<year>2013</year>). &#x201C;<article-title>Active learning for dimensional speech emotion recognition</article-title>&#x201D; in <source>Proceedings of Interspeech</source>. eds. <person-group person-group-type="editor"><name><surname>Bimbot</surname> <given-names>F.</given-names></name> <name><surname>Cerisara</surname> <given-names>C.</given-names></name> <name><surname>Fougeron</surname> <given-names>C.</given-names></name> <name><surname>Gravier</surname> <given-names>G.</given-names></name> <name><surname>Lamel</surname> <given-names>L.</given-names></name> <name><surname>Pellegrino</surname> <given-names>F.</given-names></name> <etal/></person-group>. <publisher-loc>Lyon, France</publisher-loc>: <publisher-name>International Speech Communication Association (ISCA)</publisher-name>. <fpage>2841</fpage>&#x2013;<lpage>2845</lpage>.</mixed-citation></ref>
<ref id="ref15"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hansen</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>Y. P.</given-names></name> <name><surname>Wolf</surname> <given-names>D.</given-names></name> <name><surname>Sechidis</surname> <given-names>K.</given-names></name> <name><surname>Ladegaard</surname> <given-names>N.</given-names></name> <name><surname>Fusaroli</surname> <given-names>R.</given-names></name></person-group> (<year>2022</year>). <article-title>A generalizable speech emotion recognition model reveals depression and remission</article-title>. <source>Acta Psychiatr. Scand.</source> <volume>145</volume>, <fpage>186</fpage>&#x2013;<lpage>199</lpage>. doi: <pub-id pub-id-type="doi">10.1111/acps.13388</pub-id>, PMID: <pub-id pub-id-type="pmid">34850386</pub-id></mixed-citation></ref>
<ref id="ref16"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Harati</surname> <given-names>S.</given-names></name> <name><surname>Crowell</surname> <given-names>A.</given-names></name> <name><surname>Mayberg</surname> <given-names>H.</given-names></name> <name><surname>Nemati</surname> <given-names>S.</given-names></name></person-group> (<year>2018</year>). &#x201C;Depression severity classification from speech emotion.&#x201D; In <italic>2018 40th Annual International Conference of the IEEE Engineering in Medicine and Biology Society (EMBC)</italic> (pp. 5763&#x2013;5766). IEEE.</mixed-citation></ref>
<ref id="ref17"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hashem</surname> <given-names>A.</given-names></name> <name><surname>Arif</surname> <given-names>M.</given-names></name> <name><surname>Alghamdi</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Speech emotion recognition approaches: a systematic review</article-title>. <source>Speech Comm.</source> <volume>154</volume>:<fpage>102974</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.specom.2023.102974</pub-id></mixed-citation></ref>
<ref id="ref18"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hashim</surname> <given-names>N. W.</given-names></name> <name><surname>Wilkes</surname> <given-names>M.</given-names></name> <name><surname>Salomon</surname> <given-names>R.</given-names></name> <name><surname>Meggs</surname> <given-names>J.</given-names></name> <name><surname>France</surname> <given-names>D. J.</given-names></name></person-group> (<year>2017</year>). <article-title>Evaluation of voice acoustics as predictors of clinical depression scores</article-title>. <source>J. Voice</source> <volume>31</volume>, <fpage>256.e1</fpage>&#x2013;<lpage>256.e6</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jvoice.2016.06.006</pub-id>, PMID: <pub-id pub-id-type="pmid">27473933</pub-id></mixed-citation></ref>
<ref id="ref19"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>F.</given-names></name> <name><surname>Fu</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Xiong</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>Automatic detection of negative symptoms in schizophrenia via acoustically measured features associated with affective flattening</article-title>. <source>IEEE Trans. Autom. Sci. Eng.</source> <volume>18</volume>, <fpage>586</fpage>&#x2013;<lpage>602</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TASE.2020.3022037</pub-id></mixed-citation></ref>
<ref id="ref20"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>B.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Detecting depression using an ensemble logistic regression model based on multiple speech features</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2018</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1155/2018/6508319</pub-id>, PMID: <pub-id pub-id-type="pmid">30344616</pub-id></mixed-citation></ref>
<ref id="ref21"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jordan</surname> <given-names>E.</given-names></name> <name><surname>Terrisse</surname> <given-names>R.</given-names></name> <name><surname>Lucarini</surname> <given-names>V.</given-names></name> <name><surname>Alrahabi</surname> <given-names>M.</given-names></name> <name><surname>Krebs</surname> <given-names>M. O.</given-names></name> <name><surname>Descl&#x00E9;s</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Speech emotion recognition in mental health: systematic review of voice-based applications</article-title>. <source>JMIR Mental Health</source> <volume>12</volume>:<fpage>e74260</fpage>. doi: <pub-id pub-id-type="doi">10.2196/74260</pub-id>, PMID: <pub-id pub-id-type="pmid">41027025</pub-id></mixed-citation></ref>
<ref id="ref22"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Kerkeni</surname> <given-names>L.</given-names></name> <name><surname>Serrestou</surname> <given-names>Y.</given-names></name> <name><surname>Mbarki</surname> <given-names>M.</given-names></name> <name><surname>Raoof</surname> <given-names>K.</given-names></name> <name><surname>Mahjoub</surname> <given-names>M. A.</given-names></name> <name><surname>Cleder</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Automatic speech emotion recognition using machine learning</article-title>&#x201D; in <source>Social media and machine learning [working title]</source>. ed. <person-group person-group-type="editor"><name><surname>Karray</surname> <given-names>F.</given-names></name></person-group>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>IntechOpen</publisher-name>.</mixed-citation></ref>
<ref id="ref23"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khorram</surname> <given-names>S.</given-names></name> <name><surname>Jaiswal</surname> <given-names>M.</given-names></name> <name><surname>Gideon</surname> <given-names>J.</given-names></name> <name><surname>McInnis</surname> <given-names>M.</given-names></name> <name><surname>Provost</surname> <given-names>E. M.</given-names></name></person-group> (<year>2018</year>). <article-title>The priori emotion dataset: linking mood to emotion detected in-the-wild</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1806.10658</pub-id></mixed-citation></ref>
<ref id="ref24"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Koops</surname> <given-names>S.</given-names></name> <name><surname>Brederoo</surname> <given-names>S. G.</given-names></name> <name><surname>de Boer</surname> <given-names>J. N.</given-names></name> <name><surname>Nadema</surname> <given-names>F. G.</given-names></name> <name><surname>Voppel</surname> <given-names>A. E.</given-names></name> <name><surname>Sommer</surname> <given-names>I. E.</given-names></name></person-group> (<year>2023</year>). <article-title>Speech as a biomarker for depression</article-title>. <source>CNS &#x0026; Neurolog. Disorders</source> <volume>22</volume>, <fpage>152</fpage>&#x2013;<lpage>160</lpage>. doi: <pub-id pub-id-type="doi">10.2174/1871527320666211213125847</pub-id>, PMID: <pub-id pub-id-type="pmid">34961469</pub-id></mixed-citation></ref>
<ref id="ref25"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lalitha</surname> <given-names>S.</given-names></name> <name><surname>Gupta</surname> <given-names>D.</given-names></name> <name><surname>Zakariah</surname> <given-names>M.</given-names></name> <name><surname>Alotaibi</surname> <given-names>Y. A.</given-names></name></person-group> (<year>2021</year>). <article-title>Mental illness disorder diagnosis using emotion variation detection from continuous English speech</article-title>. <source>Comput. Mater. Contin.</source> <volume>69</volume>, <fpage>3217</fpage>&#x2013;<lpage>3238</lpage>. doi: <pub-id pub-id-type="doi">10.32604/cmc.2021.018406</pub-id></mixed-citation></ref>
<ref id="ref26"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Lalitha</surname> <given-names>S.</given-names></name> <name><surname>Tripathi</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). &#x201C;Emotion detection using perceptual based speech features.&#x201D; In <italic>2016 IEEE annual India conference (INDICON)</italic> (pp. 1&#x2013;5). IEEE.</mixed-citation></ref>
<ref id="ref27"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>S.</given-names></name> <name><surname>Suh</surname> <given-names>S. W.</given-names></name> <name><surname>Kim</surname> <given-names>T.</given-names></name> <name><surname>Kim</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>K. H.</given-names></name> <name><surname>Lee</surname> <given-names>J. R.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Screening major depressive disorder using vocal acoustic features in the elderly by sex</article-title>. <source>J. Affect. Disord.</source> <volume>291</volume>, <fpage>15</fpage>&#x2013;<lpage>23</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jad.2021.04.098</pub-id>, PMID: <pub-id pub-id-type="pmid">34022551</pub-id></mixed-citation></ref>
<ref id="ref28"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Funakoshi</surname> <given-names>K.</given-names></name> <name><surname>Okumura</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). &#x201C;After: active learning based fine-tuning framework for speech emotion recognition.&#x201D; In <italic>2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)</italic> (pp. 1&#x2013;8). IEEE.</mixed-citation></ref>
<ref id="ref29"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Low</surname> <given-names>D. M.</given-names></name> <name><surname>Bentley</surname> <given-names>K. H.</given-names></name> <name><surname>Ghosh</surname> <given-names>S. S.</given-names></name></person-group> (<year>2020</year>). <article-title>Automated assessment of psychiatric disorders using speech: a systematic review</article-title>. <source>Laryngoscope Investigative Otolaryngol.</source> <volume>5</volume>, <fpage>96</fpage>&#x2013;<lpage>116</lpage>. doi: <pub-id pub-id-type="doi">10.1002/lio2.354</pub-id>, PMID: <pub-id pub-id-type="pmid">32128436</pub-id></mixed-citation></ref>
<ref id="ref30"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Manchia</surname> <given-names>M.</given-names></name> <name><surname>Pisanu</surname> <given-names>C.</given-names></name> <name><surname>Squassina</surname> <given-names>A.</given-names></name> <name><surname>Carpiniello</surname> <given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>Challenges and future prospects of precision medicine in psychiatry</article-title>. <source>Pharmacogenomics Personalized Med.</source> <volume>13</volume>, <fpage>127</fpage>&#x2013;<lpage>140</lpage>. doi: <pub-id pub-id-type="doi">10.2147/PGPM.S198225</pub-id>, PMID: <pub-id pub-id-type="pmid">32425581</pub-id></mixed-citation></ref>
<ref id="ref31"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Mar</surname> <given-names>L. L.</given-names></name> <name><surname>Pa</surname> <given-names>W. P.</given-names></name></person-group> (<year>2019</year>). Depression detection from speech emotion recognition (Doctoral dissertation, MERAL Portal).</mixed-citation></ref>
<ref id="ref32"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Marmor</surname> <given-names>S.</given-names></name> <name><surname>Horvath</surname> <given-names>K. J.</given-names></name> <name><surname>Lim</surname> <given-names>K. O.</given-names></name> <name><surname>Misono</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>Voice problems and depression among adults in the U nited S tates</article-title>. <source>Laryngoscope</source> <volume>126</volume>, <fpage>1859</fpage>&#x2013;<lpage>1864</lpage>. doi: <pub-id pub-id-type="doi">10.1002/lary.25819</pub-id>, PMID: <pub-id pub-id-type="pmid">26691195</pub-id></mixed-citation></ref>
<ref id="ref33"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mart&#x00ED;nez-S&#x00E1;nchez</surname> <given-names>F.</given-names></name> <name><surname>Muela-Mart&#x00ED;nez</surname> <given-names>J. A.</given-names></name> <name><surname>Cort&#x00E9;s-Soto</surname> <given-names>P.</given-names></name> <name><surname>Meil&#x00E1;n</surname> <given-names>J. J. G.</given-names></name> <name><surname>Ferr&#x00E1;ndiz</surname> <given-names>J. A. V.</given-names></name> <name><surname>Caparr&#x00F3;s</surname> <given-names>A. E.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Can the acoustic analysis of expressive prosody discriminate schizophrenia?</article-title> <source>Span. J. Psychol.</source> <volume>18</volume>:<fpage>E86</fpage>. doi: <pub-id pub-id-type="doi">10.1017/sjp.2015.85</pub-id></mixed-citation></ref>
<ref id="ref34"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>W.</given-names></name> <name><surname>Flint</surname> <given-names>J.</given-names></name> <name><surname>Shenhav</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Hu</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Re-examining the robustness of voice features in predicting depression: compared with baseline of confounders</article-title>. <source>PLoS One</source> <volume>14</volume>:<fpage>e0218172</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0218172</pub-id>, PMID: <pub-id pub-id-type="pmid">31220113</pub-id></mixed-citation></ref>
<ref id="ref35"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parola</surname> <given-names>A.</given-names></name> <name><surname>Simonsen</surname> <given-names>A.</given-names></name> <name><surname>Bliksted</surname> <given-names>V.</given-names></name> <name><surname>Fusaroli</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>T138. acoustic patterns in schizophrenia: a systematic review and meta-analysis</article-title>. <source>Schizophr. Bull.</source> <volume>44</volume>:<fpage>S169</fpage>. doi: <pub-id pub-id-type="doi">10.1093/schbul/sby016.415</pub-id></mixed-citation></ref>
<ref id="ref36"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pathan</surname> <given-names>H. B.</given-names></name> <name><surname>Preeth</surname> <given-names>S.</given-names></name> <name><surname>Bhavsingh</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Revolutionizing PTSD detection and emotion recognition through novel speech-based machine and deep learning algorithms</article-title>. <source>Front. Collaborative Res.</source> <volume>1</volume>, <fpage>35</fpage>&#x2013;<lpage>44</lpage>.</mixed-citation></ref>
<ref id="ref37"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Patil</surname> <given-names>M.</given-names></name> <name><surname>Wadhai</surname> <given-names>V.</given-names></name></person-group> (<year>2021</year>). &#x201C;Selection of classifiers for depression detection using acoustic features.&#x201D; In <italic>2021 International Conference on Computational Intelligence and Computing Applications (ICCICA)</italic> (pp. 1&#x2013;4). IEEE.</mixed-citation></ref>
<ref id="ref38"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rapcan</surname> <given-names>V.</given-names></name> <name><surname>D&#x2019;Arcy</surname> <given-names>S.</given-names></name> <name><surname>Yeap</surname> <given-names>S.</given-names></name> <name><surname>Afzal</surname> <given-names>N.</given-names></name> <name><surname>Thakore</surname> <given-names>J.</given-names></name> <name><surname>Reilly</surname> <given-names>R. B.</given-names></name></person-group> (<year>2010</year>). <article-title>Acoustic and temporal analysis of speech: a potential biomarker for schizophrenia</article-title>. <source>Med. Eng. Phys.</source> <volume>32</volume>, <fpage>1074</fpage>&#x2013;<lpage>1079</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.medengphy.2010.07.013</pub-id>, PMID: <pub-id pub-id-type="pmid">20692864</pub-id></mixed-citation></ref>
<ref id="ref39"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rejaibi</surname> <given-names>E.</given-names></name> <name><surname>Komaty</surname> <given-names>A.</given-names></name> <name><surname>Meriaudeau</surname> <given-names>F.</given-names></name> <name><surname>Agrebi</surname> <given-names>S.</given-names></name> <name><surname>Othmani</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>MFCC-based recurrent neural network for automatic clinical depression recognition and assessment from speech</article-title>. <source>Biomed. Signal Process. Control.</source> <volume>71</volume>:<fpage>103107</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bspc.2021.103107</pub-id></mixed-citation></ref>
<ref id="ref40"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Scherer</surname> <given-names>S.</given-names></name> <name><surname>Lucas</surname> <given-names>G. M.</given-names></name> <name><surname>Gratch</surname> <given-names>J.</given-names></name> <name><surname>Rizzo</surname> <given-names>A. S.</given-names></name> <name><surname>Morency</surname> <given-names>L. P.</given-names></name></person-group> (<year>2015</year>). <article-title>Self-reported symptoms of depression and PTSD are associated with reduced vowel space in screening interviews</article-title>. <source>IEEE Trans. Affect. Comput.</source> <volume>7</volume>, <fpage>59</fpage>&#x2013;<lpage>73</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TAFFC.2015.2440264</pub-id></mixed-citation></ref>
<ref id="ref41"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siena</surname> <given-names>F. L.</given-names></name> <name><surname>Vernon</surname> <given-names>M.</given-names></name> <name><surname>Watts</surname> <given-names>P.</given-names></name> <name><surname>Byrom</surname> <given-names>B.</given-names></name> <name><surname>Crundall</surname> <given-names>D.</given-names></name> <name><surname>Breedon</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>Proof-of-concept study: a mobile application to derive clinical outcome measures from expression and speech for mental health status evaluation</article-title>. <source>J. Med. Syst.</source> <volume>44</volume>:<fpage>209</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s10916-020-01671-x</pub-id>, PMID: <pub-id pub-id-type="pmid">33175234</pub-id></mixed-citation></ref>
<ref id="ref42"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Stasak</surname> <given-names>B.</given-names></name> <name><surname>Epps</surname> <given-names>J.</given-names></name> <name><surname>Cummins</surname> <given-names>N.</given-names></name> <name><surname>Goecke</surname> <given-names>R.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>An investigation of emotional speech in depression classification</article-title>&#x201D; in <source>Proceedings of Interspeech</source>. eds. <person-group person-group-type="editor"><name><surname>Nelson</surname> <given-names>M.</given-names></name> <name><surname>Hynek</surname> <given-names>H.</given-names></name> <name><surname>Tony</surname> <given-names>H.</given-names></name></person-group>. <publisher-loc>San Francisco, USA</publisher-loc>: <publisher-name>International Speech Communication Association (ISCA)</publisher-name>. <fpage>485</fpage>&#x2013;<lpage>489</lpage>.</mixed-citation></ref>
<ref id="ref43"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stolar</surname> <given-names>M. N.</given-names></name> <name><surname>Lech</surname> <given-names>M.</given-names></name> <name><surname>Stolar</surname> <given-names>S. J.</given-names></name> <name><surname>Allen</surname> <given-names>N. B.</given-names></name></person-group> (<year>2018</year>). <article-title>Detection of adolescent depression from speech using optimised spectral roll-off parameters</article-title>. <source>Biom. J.</source> <volume>2</volume>, <fpage>2574</fpage>&#x2013;<lpage>1241</lpage>. doi: <pub-id pub-id-type="doi">10.26717/BJSTR.2018.05.001156</pub-id></mixed-citation></ref>
<ref id="ref44"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tahir</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Chakraborty</surname> <given-names>D.</given-names></name> <name><surname>Thalmann</surname> <given-names>N.</given-names></name> <name><surname>Thalmann</surname> <given-names>D.</given-names></name> <name><surname>Maniam</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Non-verbal speech cues as objective measures for negative symptoms in patients with schizophrenia</article-title>. <source>PLoS One</source> <volume>14</volume>:<fpage>e0214314</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0214314</pub-id>, PMID: <pub-id pub-id-type="pmid">30964869</pub-id></mixed-citation></ref>
<ref id="ref45"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Teixeira</surname> <given-names>F. L.</given-names></name> <name><surname>Costa</surname> <given-names>M. R. E.</given-names></name> <name><surname>Abreu</surname> <given-names>J. P.</given-names></name> <name><surname>Cabral</surname> <given-names>M.</given-names></name> <name><surname>Soares</surname> <given-names>S. P.</given-names></name> <name><surname>Teixeira</surname> <given-names>J. P.</given-names></name></person-group> (<year>2023</year>). <article-title>A narrative review of speech and EEG features for schizophrenia detection: progress and challenges</article-title>. <source>Bioengineering</source> <volume>10</volume>:<fpage>493</fpage>. doi: <pub-id pub-id-type="doi">10.3390/bioengineering10040493</pub-id>, PMID: <pub-id pub-id-type="pmid">37106680</pub-id></mixed-citation></ref>
<ref id="ref46"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tricco</surname> <given-names>A. C.</given-names></name> <name><surname>Lillie</surname> <given-names>E.</given-names></name> <name><surname>Zarin</surname> <given-names>W.</given-names></name> <name><surname>O&#x2019;Brien</surname> <given-names>K. K.</given-names></name> <name><surname>Colquhoun</surname> <given-names>H.</given-names></name> <name><surname>Levac</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>PRISMA extension for scoping reviews (PRISMA-ScR): checklist and explanation</article-title>. <source>Ann. Intern. Med.</source> <volume>169</volume>, <fpage>467</fpage>&#x2013;<lpage>473</lpage>. doi: <pub-id pub-id-type="doi">10.7326/M18-0850</pub-id></mixed-citation></ref>
<ref id="ref47"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Valstar</surname> <given-names>M.</given-names></name> <name><surname>Schuller</surname> <given-names>B.</given-names></name> <name><surname>Smith</surname> <given-names>K.</given-names></name> <name><surname>Eyben</surname> <given-names>F.</given-names></name></person-group>, (<year>2013</year>). &#x201C;Avec 2013: the continuous audio/visual emotion and depression recognition challenge.&#x201D; In <italic>Proceedings of the 3rd ACM international workshop on Audio/visual emotion challenge</italic>. <publisher-loc>Barcelona, Spain</publisher-loc>: <publisher-name>Proceedings published by ACM</publisher-name>. (pp. 3&#x2013;10).</mixed-citation></ref>
<ref id="ref48"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wanderley Espinola</surname> <given-names>C.</given-names></name> <name><surname>Gomes</surname> <given-names>J. C.</given-names></name> <name><surname>M&#x00F4;nica Silva Pereira</surname> <given-names>J.</given-names></name> <name><surname>dos Santos</surname> <given-names>W. P.</given-names></name></person-group> (<year>2022</year>). <article-title>Detection of major depressive disorder, bipolar disorder, schizophrenia and generalized anxiety disorder using vocal acoustic analysis and machine learning: an exploratory study</article-title>. <source>Res. Biomedical Eng.</source> <volume>38</volume>, <fpage>813</fpage>&#x2013;<lpage>829</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s42600-022-00222-2</pub-id></mixed-citation></ref>
<ref id="ref49"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Qi</surname> <given-names>W.</given-names></name> <name><surname>Su</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Zhou</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>), &#x201C;A novel end-to-end speech emotion recognition network with stacked transformer layers.&#x201D; In <italic>ICASSP 2021&#x2013;2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</italic> (pp. 6289&#x2013;6293). IEEE.</mixed-citation></ref>
<ref id="ref51"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Folarin</surname> <given-names>A. A.</given-names></name> <name><surname>Dineley</surname> <given-names>J.</given-names></name> <name><surname>Conde</surname> <given-names>P.</given-names></name> <name><surname>de Angel</surname> <given-names>V.</given-names></name> <name><surname>Sun</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Identifying depression-related topics in smartphone-collected free-response speech recordings using an automatic speech recognition system and a deep learning topic model</article-title>. <source>J. Affect. Disord.</source> <volume>355</volume>, <fpage>40</fpage>&#x2013;<lpage>49</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jad.2024.03.106</pub-id>, PMID: <pub-id pub-id-type="pmid">38552911</pub-id></mixed-citation></ref>
</ref-list><fn-group><fn id="fn0001" fn-type="custom" custom-type="edited-by"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1364632/overview">Panagiotis Tzirakis</ext-link>, Hume AI, United States</p></fn>
<fn id="fn0002" fn-type="custom" custom-type="reviewed-by"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1172317/overview">Birger Moell</ext-link>, KTH Royal Institute of Technology, Sweden; <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2856509/overview">Dongyuan Li</ext-link>, The University of Tokyo, Japan</p></fn></fn-group></back>
</article>