<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychiatry</journal-id>
<journal-title>Frontiers in Psychiatry</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychiatry</abbrev-journal-title>
<issn pub-type="epub">1664-0640</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyt.2025.1595197</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Psychiatry</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Deeper insight into speech characteristics of patients at ultra-high risk using classification and explainability models</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kim-Dufor</surname>
<given-names>Deok-Hee</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2302859/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Walter</surname>
<given-names>Michel</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Krebs</surname>
<given-names>Marie-Odile</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Haralambous</surname>
<given-names>Yannis</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lenca</surname>
<given-names>Philippe</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lemey</surname>
<given-names>Christophe</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/783905/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Limics, Sorbonne Universit&#xe9;, Universit&#xe9; Sorbonne Paris-Nord, INSERM</institution>, <addr-line>Paris</addr-line>,&#xa0;<country>France</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Unit&#xe9; de Recherche Clinique en Psychiatrie (URCP), Department of Psychiatry, Centre Hospitalier Universitaire (CHU) de Brest</institution>, <addr-line>Brest</addr-line>,&#xa0;<country>France</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>University of Paris, Groupe Hospitalier Universitaire de Paris (GHU)-Paris, Service Hospitalo-Universitaire, Sainte-Anne, Centre d'&#xe9;valuation pour Jeunes Adultes et ADolescents (C&#x2019;JAAD)</institution>, <addr-line>Paris</addr-line>,&#xa0;<country>France</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>IMT Atlantique, Lab-STICC, UMR CNRS 6285</institution>, <addr-line>Brest</addr-line>,&#xa0;<country>France</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Donn&#xe9;es, Mod&#xe8;les, Informations &amp; D&#xe9;cisions (DECIDE), Department of LUSSI, Institut Mines-T&#xe9;l&#xe9;com (IMT) Atlantique</institution>, <addr-line>Brest</addr-line>,&#xa0;<country>France</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Consultation d&#x2019;Evaluation de la VUln&#xe9;rabilit&#xe9; Psychologique (CEVUP), Department of Psychiatry, CHU de Brest</institution>, <addr-line>Brest</addr-line>,&#xa0;<country>France</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Saturnino Luz, University of Edinburgh, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Sofia De La Fuente Garcia, University of Edinburgh, United Kingdom</p>
<p>Bahman Mirheidari, The University of Sheffield, United Kingdom</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Deok-Hee Kim-Dufor, <email xlink:href="mailto:dh.kimdufor@gmail.com">dh.kimdufor@gmail.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>06</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1595197</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>03</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>05</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Kim-Dufor, Walter, Krebs, Haralambous, Lenca and Lemey</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Kim-Dufor, Walter, Krebs, Haralambous, Lenca and Lemey</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Peculiar use of language and even language deficits are one of the well-known signs of schizophrenia. Different language features analyzed using natural language processing and machine learning have been reported to differentiate patients at ultra-high risk for psychosis. However, it has not always been explained how, and to what extent, those linguistic markers allow the distinction of patients. This study aims to find relevant linguistic markers for classifying patients at ultra-high risk and explain how the detected markers contribute to the classification.</p>
</sec>
<sec>
<title>Methods</title>
<p>The first consultations with a psychiatrist of 68 patients (15 not-at-risk patients, 45 at-risk patients, and 8 patients with first episode psychosis) were recorded, transcribed verbatim, and annotated for analyses using natural language processing. A gradient-boosted decision tree algorithm was tested to evaluate its potential to correctly classify three categories of patients and find relevant linguistic markers at the level of lexical richness, semantic coherence, speech disfluency, and syntactic complexity. The Synthetic Minority Oversampling Technique was used to handle imbalanced data, and the SHapley Additive exPlanations (SHAP) values were computed to measure feature importance and each feature&#x2019;s contributions to the classification.</p>
</sec>
<sec>
<title>Results</title>
<p>The model yielded good performance, that is, 0.82 accuracy, 0.82 F2-score, 0.85 precision, 0.82 recall, and 0.86 ROC&#x2013;AUC score, with four linguistic variables that concern weak coherence, the use of &#x201c;I,&#x201d; and filled pauses.</p>
</sec>
<sec>
<title>Discussion</title>
<p>The findings in this study suggest that weak coherence play a key role in classification. No significant differences in the use of &#x201c;I&#x201d; and filled pauses were found between groups using a statistical test, but an explainability model showed its different contributions. The contribution of each linguistic feature to the classification by patient group provided deeper insight into linguistic manifestations of each patient group and their subtle differences, which could help better analyze and understand patients&#x2019; language behaviors.</p>
</sec>
</abstract>
<kwd-group>
<kwd>UHR patients</kwd>
<kwd>spoken language</kwd>
<kwd>natural language processing</kwd>
<kwd>XGBoost</kwd>
<kwd>SMOTE</kwd>
<kwd>SHAP values</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="120"/>
<page-count count="15"/>
<word-count count="5897"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Schizophrenia</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>People with schizophrenia present with significant impairments stemming from disordered cognitive functioning (<xref ref-type="bibr" rid="B1">1</xref>). This mental illness manifests itself in characteristic symptoms such as delusions, hallucinations, disorganized thinking and behaviors, limited speech and expression of emotions, and social withdrawal. Early detection and treatment of schizophrenia have been proven to lead patients to favorable prognosis and better quality of life (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B3">3</xref>). They could indeed reduce the risks and disorders associated with the first symptoms by engaging patients who present with prodromal symptoms in a care pathway (<xref ref-type="bibr" rid="B4">4</xref>) and limit the duration of untreated psychosis (DUP) by means of a treatment at the onset of the first episode of psychosis (FEP). The DUP is one of the key prognostic factors both in FEP (<xref ref-type="bibr" rid="B5">5</xref>) and in chronic schizophrenia (<xref ref-type="bibr" rid="B6">6</xref>). Different clinical assessments allow prodromal symptoms to be identified such as the Comprehensive Assessment of At-Risk Mental States (CAARMS), the Structured Interview of Psychosis-risk Syndromes (SIPS) from the &#x201c;Ultra-High Risk (UHR)&#x201d; criteria, and the Schizophrenia Proneness Instrument&#x2014;Adult (SPI-A) from the basic symptom concept. Even though these tools show acceptable or fairly good performances, they still have a somewhat limited rate of prediction (<xref ref-type="bibr" rid="B7">7</xref>). Complementary elements for better predictions have therefore become a desideratum, and natural language processing (NLP) comes into play. Peculiar uses of language in schizophrenia (<xref ref-type="bibr" rid="B8">8</xref>&#x2013;<xref ref-type="bibr" rid="B10">10</xref>) have been reported in the literature and are one of the well-known signs (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>). They are very easily noticeable and even qualified as &#x201c;schizophrenic language&#x201d; and &#x201c;schizophrenese&#x201d; by some authors in the last century (<xref ref-type="bibr" rid="B13">13</xref>&#x2013;<xref ref-type="bibr" rid="B16">16</xref>). Peculiarities are observed at different language levels ranging from words to sentence structure, coherence, pragmatics (<xref ref-type="bibr" rid="B17">17</xref>&#x2013;<xref ref-type="bibr" rid="B21">21</xref>) as itemized in the Scale for the Assessment of Thought, Language, and Communication by Andreasen like neologism, word approximation, poverty of speech, poverty of content, tangentiality, derailment, incoherence, and stilted speech (<xref ref-type="bibr" rid="B8">8</xref>). Based on the idea that self-disturbance is one of the core features of schizophrenia, a phenomenological approach to the sense of self in patients has developed (<xref ref-type="bibr" rid="B22">22</xref>&#x2013;<xref ref-type="bibr" rid="B24">24</xref>) along with studies on the use of first-person pronouns (<xref ref-type="bibr" rid="B25">25</xref>&#x2013;<xref ref-type="bibr" rid="B29">29</xref>). Language analysis of syntactic variables was already proposed in the 1980s as a potential diagnostic aid (<xref ref-type="bibr" rid="B30">30</xref>&#x2013;<xref ref-type="bibr" rid="B32">32</xref>), since differences were observed between schizophrenics, maniacs, and controls (<xref ref-type="bibr" rid="B30">30</xref>, <xref ref-type="bibr" rid="B31">31</xref>). Even though language analyses turned out to have great potential, they were highly time consuming and likely to be subjective because they had to be manually carried out. Automated language analyses are more objective methods and unlimited in data size. Many studies have therefore explored language in schizophrenia and searched for linguistic markers to be used as a diagnostic aid along with biomarkers such as brain imaging, genetic testing, and blood tests (<xref ref-type="bibr" rid="B33">33</xref>&#x2013;<xref ref-type="bibr" rid="B35">35</xref>). With the development of artificial intelligence, analysis techniques, such as NLP and machine learning (ML) models, have become more sophisticated and yielded more propitious results. These techniques have been used on linguistic data in a growing number of studies on mental health (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B37">37</xref>), namely, those on schizophrenia and FEP (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>): latent semantic analysis for quantifying speech coherence (<xref ref-type="bibr" rid="B40">40</xref>), semantic, lexical, and pragmatic features (<xref ref-type="bibr" rid="B41">41</xref>&#x2013;<xref ref-type="bibr" rid="B44">44</xref>), speech graph connectivity for measuring thought disorder in schizophrenia and mania (<xref ref-type="bibr" rid="B45">45</xref>, <xref ref-type="bibr" rid="B46">46</xref>) and for predicting transition (<xref ref-type="bibr" rid="B47">47</xref>, <xref ref-type="bibr" rid="B48">48</xref>), longitudinal classification of FEP (<xref ref-type="bibr" rid="B49">49</xref>), clustering for constructing language profiles of heterogeneous linguistic behaviors of patients with schizophrenia for early intervention (<xref ref-type="bibr" rid="B50">50</xref>) and prognosis (<xref ref-type="bibr" rid="B51">51</xref>), and a combination of acoustic and semantic features for classifying schizophrenia-spectrum disorders (<xref ref-type="bibr" rid="B52">52</xref>), to name a few. The aims of this exploratory study were to detect relevant language features that could classify patients by their status at their first consultation with a psychiatrist and seek to explain classification results with respect to clinical observations. Among the linguistic markers found in these studies (<xref ref-type="bibr" rid="B40">40</xref>&#x2013;<xref ref-type="bibr" rid="B51">51</xref>), the most frequent language feature is semantic coherence despite different types and lengths of corpus. It was therefore hypothesized that semantic coherence would be part of the relevant linguistic markers in conversational discourses of patients at ultra-high risk. With the disturbed sense of self observed in the clinic, it was also hypothesized that the use of first-person singular pronoun would vary depending on the UHR patient groups.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Participants</title>
<p>Sixty-eight patients (34 males, 34 females; mean age = 19.3 &#xb1; 2.86) participated in the present study. Out of the 68 patients, 15 were assessed as NAR (7 males, 8 females; mean age = 19.5 &#xb1; 2.24), 45 as AR (22 males, 23 females; mean age = 19.2 &#xb1; 2.83), and 8 as FEP (5 males, 3 females; mean age = 19.7 &#xb1; 3.78) using the CAARMS at T0. In total, 33 patients had antidepressants and/or anxiolytics, 5 were under neuroleptic treatment for less than 6 months, and 20 had no drug treatment. Healthy controls were not recruited separately to respect the same conditions of collecting data for each of the three groups, that is, a consultation with a psychiatrist. All were native speakers of French with an IQ superior to 70 and were informed of the study. Education levels were as follows: NAR [years of education (YoE) = 12.07 &#xb1; 1.34], AR (YoE = 11.58 &#xb1; 1.32), and FEP (YoE = 12 &#xb1; 1.73). A statement of non-opposition to the study was signed by their physician or the parents of underage patients.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Collection of patients&#x2019; speech and transcription</title>
<p>The recruited patients were recorded during their first consultations with a psychiatrist at the Center for Evaluation of Psychological Vulnerability (CEVUP) of the University Hospital of Brest, France. The first consultation with a psychiatrist is the starting point of the care pathway at the CEVUP. It is therefore labeled T0 (time zero), and a 2-year follow-up is indicated as T2. The interviews are semi-structured with some predetermined questions on the patient&#x2019;s problems. The topics broached are the patient&#x2019;s background, family, social relationships, socio-professional insertion, complaints about their symptoms, and any other topics based on what is said by the patient. Some additional questions are asked if more detailed information is needed for better understanding of the help seeker&#x2019;s problems to assess their risk for psychosis. The transcripts have a conversational form between a psychiatrist and a patient. A nurse participated in the consultations, but she seldom spoke, and even when she did, it was only to provide the patient with supplementary information on the care pathway at the end of the consultations. The total duration of each recording is approximately 1 h. The mean total number of all words is 4,979.18 (SD = 2,448.70). The entire utterances including filled pauses, neologisms, and mispronunciations were transcribed verbatim using Microsoft Word by two trained assistants with clear instructions. Each speech turn starts on a new line and that of the healthcare provider is marked with an octothorpe (#) at the beginning and at the end. The present study has been approved by the IRB&#x2014;Comit&#xe9; de Protection des Personnes EST-III (CPP:18.04.03, ID-RCB: 2017-A02702-51).</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Preprocessing</title>
<p>An experienced linguist carried out preprocessing following predefined instructions. The spellings were manually double checked and corrected in all the transcripts without affecting their verbatim nature. Three different symbols, inspired by the method proposed by Foster and colleagues (<xref ref-type="bibr" rid="B53">53</xref>), were used to mark the elements required for analyses as follows:</p>
<list list-type="bullet">
<list-item>
<p>
<bold>{}</bold> for speech disfluency such as filled pause, repetition, false start, auto-correction, and auto-interruption/abandonment</p>
</list-item>
<list-item>
<p>
<bold>|</bold> for clauses whose nucleus is a conjugated verb</p>
</list-item>
<list-item>
<p>
<bold>&lt; &gt;</bold> for minor utterances (no conjugated verbs).</p>
</list-item>
</list>
<p>The transcripts were segmented in three ways: each speech turn as a segment, each sentence as a segment, and each sentence without the healthcare provider&#x2019;s speech as a segment. For the first segment, each new line was a segment; for the second, each punctuation; and for the last, the whole new lines starting and ending with octothorpes were removed using Python as well as the blank lines generated by this removal process.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Linguistic variables</title>
<p>The preprocessed transcripts were analyzed using NLP techniques with Python, which resulted in 33 features at the lexical, syntactic, and semantic levels and that of speech fluency (see Table in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>).</p>
<sec id="s2_4_1">
<label>2.4.1</label>
<title>Lexical level</title>
<p>Lexical richness was measured to explore the variety of words and the quality of vocabulary. For the former, lexical diversity was calculated using the type&#x2013;token ratio (<xref ref-type="bibr" rid="B54">54</xref>). For the latter, the proportion of content words (nouns, verbs, adjectives, and adverbs) to the total number of words, called lexical density (<xref ref-type="bibr" rid="B55">55</xref>), was measured. Since function words are excluded, lexical density reflects how informative the discourse is. Disturbed self-experience and different patterns of use of the first-person singular pronoun in people with schizophrenia have been reported (<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B56">56</xref>). The use of personal pronouns was explored through three different measures as follows: the proportion of &#x201c;I&#x201d; to the total number of subject personal pronouns, the proportion of &#x201c;I&#x201d; to the total number of words, and the ratio of the first-person singular subject pronoun to the first-person object pronoun. The analyses at the lexical&#xa0;level&#xa0;were carried out on the lemmatized corpus using treetaggerwrapper (<xref ref-type="bibr" rid="B57">57</xref>).</p>
</sec>
<sec id="s2_4_2">
<label>2.4.2</label>
<title>Syntactic level</title>
<p>Syntactic complexity and poverty of speech were measured. The analyses were based on lexicogrammatical constituency in functional grammar. Constituency is the hierarchical compositional structure of language, and this hierarchy of units is denominated as a rank scale, with each step in the hierarchy referred to as one rank (<xref ref-type="bibr" rid="B58">58</xref>). The ranks of lexicogrammatical constituency are clause &gt; phrase/group &gt; word &gt; morpheme, wherein the clause is the highest unit and the central processing unit. In addition, this unit is one of the five levels in the grammatical system (<xref ref-type="bibr" rid="B59">59</xref>) and the primary unit in immediate speech processing (<xref ref-type="bibr" rid="B60">60</xref>). The clause has therefore been determined as the basic syntactic unit in this study. The utterances were segmented into clauses whose nucleus is a conjugated verb. When a group of words lacks a conjugated verb, it is considered a minor utterance. As for syntactic complexity, Szmerecs&#xe1;ny compared syntax tree-based node counts, length-based word counts, and index of syntactic complexity calculated based on subordinators and embeddedness with regard to their accuracy and applicability (<xref ref-type="bibr" rid="B61">61</xref>). The results showed that all the three methods were almost perfect proxies, and therefore the most economical method, word counts, could be used. The average number of words per clause was therefore calculated as a measure of syntactic complexity. In turn-taking between a patient and a psychiatrist, the number of the patient&#x2019;s turns was counted, and the proportion of the turns only with minor utterances (short answers) to the total number of their turns was calculated. A patient&#x2019;s turn is considered minor utterance when the patient answers with simple words such as &#x201c;yes,&#x201d; &#x201c;no,&#x201d; &#x201c;OK,&#x201d; or a group of words without developing the reply. For example, to the question &#x201c;How are you feeling today?&#x201d;, the reply would be &#x201c;so so/a little better/not really happy about all this.&#x201d; This type of utterances is in line with &#x201c;poverty of speech,&#x201d; which is widely described in the literature (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B12">12</xref>). All the disfluency elements have been removed from the corpus prior to the syntactic analyses.</p>
</sec>
<sec id="s2_4_3">
<label>2.4.3</label>
<title>Semantic level</title>
<p>Latent semantic analysis (LSA) (<xref ref-type="bibr" rid="B62">62</xref>, <xref ref-type="bibr" rid="B63">63</xref>) has been applied to measure incoherence in speech (<xref ref-type="bibr" rid="B40">40</xref>, <xref ref-type="bibr" rid="B41">41</xref>) and turned out to be fairly efficient when combined with other linguistic features (<xref ref-type="bibr" rid="B41">41</xref>&#x2013;<xref ref-type="bibr" rid="B43">43</xref>, <xref ref-type="bibr" rid="B49">49</xref>). LSA is a widely used NLP technique that analyzes texts to explore the relationships between a set of documents and the terms inside those documents. The underlying idea of LSA is that semantically similar words occur in similar texts, and thereby the cooccurrences of terms in large corpora of texts are used for measuring the lexical proximity/semantic similarity of terms of a language. LSA was chosen over other techniques for the following assets: a) the technique is based on a psychological theory of meaning and has shown results similar to human evaluations in educational applications (<xref ref-type="bibr" rid="B63">63</xref>); b) early studies using this technique paved the way for the use of NLP in early detection of psychosis (<xref ref-type="bibr" rid="B40">40</xref>, <xref ref-type="bibr" rid="B41">41</xref>, <xref ref-type="bibr" rid="B64">64</xref>, <xref ref-type="bibr" rid="B65">65</xref>); c) LSA can handle longer passages of words (<xref ref-type="bibr" rid="B66">66</xref>) and synonyms in case of word redundancy for the avoidance of repetition (<xref ref-type="bibr" rid="B63">63</xref>); and d) contrary to new transformer-based models, this technique is not sensitive to initialization parameters, which allows consistent results. In addition, an LSA-based text analysis tool called Coh-Metrix (<xref ref-type="bibr" rid="B67">67</xref>, <xref ref-type="bibr" rid="B68">68</xref>) has been efficiently used in studies on formal thought disorder (FTD) (<xref ref-type="bibr" rid="B56">56</xref>, <xref ref-type="bibr" rid="B69">69</xref>&#x2013;<xref ref-type="bibr" rid="B71">71</xref>). In the present study, semantic coherence was measured in three different types: intersubjective, subjective, and subjective without doctor (abbreviated henceforth as <italic>wodr</italic>) coherence. In the first type, semantic coherence was measured based on turn-taking, which represents dialogue coherence, inter-turn comparison; in the second, based on punctuation marks, such as periods and question marks, which could be called sentence-to-sentence coherence; and in the third, only the patients&#x2019; speech was considered. For the semantic analyses, the transcripts were not lemmatized (<xref ref-type="bibr" rid="B72">72</xref>), stop words were removed, and the disfluency elements were kept for the sake of semantic integrity.</p>
</sec>
<sec id="s2_4_4">
<label>2.4.4</label>
<title>Speech fluency</title>
<p>Speech flow can vary in any individuals depending on their situation, state of mind, and/or fatigue. Disfluencies in speech comprise unfilled pauses (silent), filled pauses (&#x201c;<italic>uh</italic>,&#x201d; &#x201c;<italic>um</italic>&#x201d;), false starts, repetitions, autocorrection, parenthetical remarks (&#x201c;<italic>well</italic>,&#x201d; &#x201c;<italic>yeah</italic>&#x201d;) (<xref ref-type="bibr" rid="B73">73</xref>), and abandoned utterances (abandonment/auto-interruption). Various features of speech disfluency in patients with psychotic disorders, such as filled pauses, autocorrection, reparandum&#x2013;interregnum repair structure, and unfilled pauses, have been studied in detail (<xref ref-type="bibr" rid="B74">74</xref>&#x2013;<xref ref-type="bibr" rid="B76">76</xref>). All the disfluency elements, except unfilled pauses, were counted, and three disfluency-related subcategories were created as features in the present study as follows: filled pauses, abandonments/auto-interruptions, and auto-corrections/repetitions/false starts. The proportion of each of the three to the total number of words was calculated. A disfluency element with several words was counted as one. Among the abandoned utterances, clauses with a subject and an incomplete predicate have constituted a variable, that is, truncated clauses.</p>
</sec>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Statistics, XGBoost Classifier, SMOTE, SHAP values</title>
<p>Statistical analyses were carried out using Python scipy (<xref ref-type="bibr" rid="B77">77</xref>) and statsmodels (<xref ref-type="bibr" rid="B78">78</xref>). Data normality was tested using Kolmogorov&#x2013;Smirnov test. For group comparisons in each of the 33 linguistic features and education levels, a Kruskal&#x2013;Wallis test and a Dunn&#x2013;Bonferroni test, as a <italic>post hoc</italic> analysis, were performed. Data homoscedasticity was verified using Levene&#x2019;s test. A Kendall&#x2019;s tau-b was calculated between the linguistic variables and the patients&#x2019; education levels as possible confounders.</p>
<p>A supervised machine learning model XGBoost, for eXtreme Gradient Boosting (<xref ref-type="bibr" rid="B79">79</xref>) was used for classification. The gradient boosting method provides higher predictive accuracy thanks to its functional characteristics, that is, it combines weak learners to give rise to a stronger learner and therefore forms a more robust model (<xref ref-type="bibr" rid="B80">80</xref>). In addition, multicollinearity does not affect the stability and robustness of the model&#x2019;s performance thanks to the capability of the algorithm to choose the best of highly correlated features (<xref ref-type="bibr" rid="B81">81</xref>). Furthermore, XGBoost has shown better performance with small datasets (<xref ref-type="bibr" rid="B82">82</xref>, <xref ref-type="bibr" rid="B83">83</xref>) than other classifiers. The dataset in the present study is imbalanced. This limitation was addressed through SMOTE (Synthetic Minority Oversampling Technique) (<xref ref-type="bibr" rid="B84">84</xref>), a statistical technique for upsampling the minority class for a better balanced dataset. This technique has already been used and proven its efficacity, for example, in diagnosis, classification, and prognosis of cancer, diabetes, and Parkinson&#x2019;s disease (<xref ref-type="bibr" rid="B85">85</xref>&#x2013;<xref ref-type="bibr" rid="B97">97</xref>) to name a few. Stratified K-fold cross validation (k = 3) was used to split the data into train and test sets, and SMOTE was subsequently conducted individually in each fold to avoid data leakage. Stratified K-fold cross validation was chosen over leave-one-out cross validation for the sake of computational time and power, and k = 3 was set considering our relatively small dataset and the number of patient groups. The test size was 0.3. Using Bayesian Optimization (<xref ref-type="bibr" rid="B98">98</xref>) to tune hyperparameters, an XGBoost Classifier was trained using the 33 features of the original data to compute the SHapley Additive exPlanation (SHAP) values (<xref ref-type="bibr" rid="B99">99</xref>), and the mean absolute SHAP values were calculated for feature selection (<xref ref-type="bibr" rid="B100">100</xref>, <xref ref-type="bibr" rid="B101">101</xref>). Another XGBoostClassifier was then trained using the outcome of feature importance based on the mean absolute SHAP values and the upsampled data. Inspired by Shapely values (<xref ref-type="bibr" rid="B102">102</xref>) from cooperative game theory, the SHAP values allow interpreting the model output by measuring the contribution of each feature to predictions. Precisely, the SHAP values reveal how much (magnitude) and either positively or negatively (direction) each feature affected the classification (<xref ref-type="bibr" rid="B99">99</xref>). This method thereby allows explanations and better interpretation of the results. The process of speech data acquisition and analyses is depicted below in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Pipeline for speech data acquisition and data analyses.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1595197-g001.tif"/>
</fig>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Statistical results</title>
<p>A Kolmogorov&#x2013;Smirnov test showed that no feature had a normal distribution (0.5 &#x2264; <italic>D</italic> &#x2264; 1 and p &lt; 0.00 in all 33 features). The results of Levene&#x2019;s test indicated homogeneity of variance in all features (p &gt; 0.05). A Kendall&#x2019;s tau-b test showed no evidence for a moderate or strong impact of years of education on the linguistic features (<italic>r<sub>&#x3c4;</sub>
</italic> = 0.24, p = 0.01 between average number of words per clause and education level; &#x2212;0.14 &#x2264; <italic>r<sub>&#x3c4;</sub>
</italic> &#x2264; 0.16, 0.07 &#x2264; p &#x2264; 0.99 in all the other pairs). A Kruskal&#x2013;Wallis test was performed on each of the 33 features of the three groups. The results revealed significant differences between the three groups in two features (intersubjective LSA minimum and subjective LSA minimum) as shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1a</bold>
</xref> (for the full table, see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>). A Dunn&#x2013;Bonferroni test was then conducted to verify which groups were different. Its results indicated significant differences either between AR and FEP or between AR and FEP, but no differences were found between NAR and AR as shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1b</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Kruskal&#x2013;Wallis test results of the main features (a) and Dunn&#x2013;Bonferroni test results (b).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" colspan="7" align="left">(a)</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="bottom" align="center">Features</th>
<th valign="bottom" align="center">Total</th>
<th valign="bottom" align="center">df</th>
<th valign="bottom" align="center">H</th>
<th valign="bottom" colspan="2" align="center">Effect size (&#x3f5;2)</th>
<th valign="bottom" align="center">p-Value</th>
</tr>
<tr>
<td valign="bottom" align="center">Intersubjective LSA median</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.4011</td>
<td valign="bottom" colspan="2" align="center">0.0358</td>
<td valign="bottom" align="center">0.3010</td>
</tr>
<tr>
<td valign="bottom" align="center">
<bold>Intersubjective LSA minimum</bold>
</td>
<td valign="bottom" align="center">
<bold>68</bold>
</td>
<td valign="bottom" align="center">
<bold>2</bold>
</td>
<td valign="bottom" align="center">
<bold>13.4282</bold>
</td>
<td valign="bottom" colspan="2" align="center">
<bold>0.2004</bold>
</td>
<td valign="bottom" align="center">
<bold>0.0012</bold>
</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA median</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.1901</td>
<td valign="bottom" colspan="2" align="center">0.0327</td>
<td valign="bottom" align="center">0.3345</td>
</tr>
<tr>
<td valign="bottom" align="center">
<bold>Subjective LSA minimum</bold>
</td>
<td valign="bottom" align="center">
<bold>68</bold>
</td>
<td valign="bottom" align="center">
<bold>2</bold>
</td>
<td valign="bottom" align="center">
<bold>8.2831</bold>
</td>
<td valign="bottom" colspan="2" align="center">
<bold>0.1236</bold>
</td>
<td valign="bottom" align="center">
<bold>0.0159</bold>
</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA wodr median</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.3885</td>
<td valign="bottom" colspan="2" align="center">0.0356</td>
<td valign="bottom" align="center">0.3029</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA wodr minimum</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">1.9917</td>
<td valign="bottom" colspan="2" align="center">0.0297</td>
<td valign="bottom" align="center">0.3694</td>
</tr>
<tr>
<td valign="bottom" align="center">Intersubjective LSA IQR</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">3.3373</td>
<td valign="bottom" colspan="2" align="center">0.0498</td>
<td valign="bottom" align="center">0.1885</td>
</tr>
<tr>
<td valign="bottom" align="center">Intersubjective LSA +1.5IQR %</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.7397</td>
<td valign="bottom" colspan="2" align="center">0.0409</td>
<td valign="bottom" align="center">0.2541</td>
</tr>
<tr>
<td valign="bottom" align="center">Intersubjective LSA &#x2212;1.5IQR %</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">4.6137</td>
<td valign="bottom" colspan="2" align="center">0.0689</td>
<td valign="bottom" align="center">0.0996</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA IQR</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">3.3373</td>
<td valign="bottom" colspan="2" align="center">0.0498</td>
<td valign="bottom" align="center">0.1885</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA +1.5IQR %</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.7397</td>
<td valign="bottom" colspan="2" align="center">0.0409</td>
<td valign="bottom" align="center">0.2541</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA &#x2212;1.5IQR %</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">4.6137</td>
<td valign="bottom" colspan="2" align="center">0.0689</td>
<td valign="bottom" align="center">0.0996</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA wodr IQR</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">3.3373</td>
<td valign="bottom" colspan="2" align="center">0.0498</td>
<td valign="bottom" align="center">0.1885</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA wodr +1.5IQR %</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.7397</td>
<td valign="bottom" colspan="2" align="center">0.0409</td>
<td valign="bottom" align="center">0.2541</td>
</tr>
<tr>
<td valign="bottom" align="center">Subjective LSA wodr &#x2212;1.5IQR %</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">4.6137</td>
<td valign="bottom" colspan="2" align="center">0.0689</td>
<td valign="bottom" align="center">0.0996</td>
</tr>
<tr>
<td valign="bottom" align="center">Lexical diversity (%)</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.7213</td>
<td valign="bottom" colspan="2" align="center">0.0406</td>
<td valign="bottom" align="center">0.2565</td>
</tr>
<tr>
<td valign="bottom" align="center">je (%)_total n</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">5.3242</td>
<td valign="bottom" colspan="2" align="center">0.0795</td>
<td valign="bottom" align="center">0.0698</td>
</tr>
<tr>
<td valign="bottom" align="center">je (%)_pp</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.4731</td>
<td valign="bottom" colspan="2" align="center">0.0369</td>
<td valign="bottom" align="center">0.2904</td>
</tr>
<tr>
<td valign="bottom" align="center">Ure's lexical density (%)</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">3.1329</td>
<td valign="bottom" colspan="2" align="center">0.0468</td>
<td valign="bottom" align="center">0.2088</td>
</tr>
<tr>
<td valign="bottom" align="center">Truncated clauses (%)</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">0.5767</td>
<td valign="bottom" colspan="2" align="center">0.0086</td>
<td valign="bottom" align="center">0.7495</td>
</tr>
<tr>
<td valign="bottom" align="center">Short answers (%)</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">2.3239</td>
<td valign="bottom" colspan="2" align="center">0.0347</td>
<td valign="bottom" align="center">0.3129</td>
</tr>
<tr>
<td valign="bottom" align="center">Ratio_subj/obj</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">1.5512</td>
<td valign="bottom" colspan="2" align="center">0.0232</td>
<td valign="bottom" align="center">0.4604</td>
</tr>
<tr>
<td valign="bottom" align="center">Filled pauses</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">4.4079</td>
<td valign="bottom" colspan="2" align="center">0.0658</td>
<td valign="bottom" align="center">0.1104</td>
</tr>
<tr>
<td valign="bottom" align="center">Abandonment</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">0.8165</td>
<td valign="bottom" colspan="2" align="center">0.0122</td>
<td valign="bottom" align="center">0.6648</td>
</tr>
<tr>
<td valign="bottom" align="center">Autocorrection_Repetition</td>
<td valign="bottom" align="center">68</td>
<td valign="bottom" align="center">2</td>
<td valign="bottom" align="center">0.5039</td>
<td valign="bottom" colspan="2" align="center">0.0075</td>
<td valign="bottom" align="center">0.7773</td>
</tr>
</tbody>
<tbody>
<tr>
<th valign="bottom" colspan="7" align="left">(b)</th>
</tr>
<tr>
<th valign="middle" align="center">Features</th>
<th valign="middle" colspan="2" align="center">NAR vs. AR (<italic>p</italic>)</th>
<th valign="middle" colspan="2" align="center">NAR vs. FEP (<italic>p</italic>)</th>
<th valign="middle" colspan="2" align="center">AR vs. FEP (<italic>p</italic>)</th>
</tr>
<tr>
<td valign="middle" align="center">Intersubjective LSA minimum</td>
<td valign="middle" colspan="2" align="center">0.1336</td>
<td valign="middle" colspan="2" align="center">
<bold>0.0007</bold>
</td>
<td valign="middle" colspan="2" align="center">
<bold>0.0267</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">Subjective LSA minimum</td>
<td valign="middle" colspan="2" align="center">1.000</td>
<td valign="middle" colspan="2" align="center">0.0696</td>
<td valign="middle" colspan="2" align="center">
<bold>0.0123</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The rows in bold are features and values with a significant difference (<italic>p</italic> &lt; 0.05).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Classification and explainability results</title>
<p>The XGBoostClassifier trained on SMOTE data with all the features yielded 0.75 accuracy, 0.73 precision, 0.75 recall, 0.74 F2-score, and 0.70 ROC&#x2013;AUC score. The most impactful features were selected based on the mean absolute values computed on the original data as shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. The first four features whose values are greater than 0.3 were selected (intersubjective LSA minimum, subjective LSA wodr minimum, the proportion of &#x201c;I&#x201d; to the total number of words, and filled pauses) for another classification using XGBoostClassifier. This cutoff selection was based on threshold tests on the first 10 features. The best result was obtained when the first four features were included; for example, with the first five features, the accuracy was slightly lower (0.79) than that with the first four features and higher than that with the whole features (0.75). The newly trained model reached 0.82 accuracy, 0.85 precision, 0.82 recall, 0.82 F2-score, and 0.86 ROC&#x2013;AUC score (see <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> for ROC&#x2013;AUC curve), and as for 95% confidence intervals (CI) of accuracy, the lower CI was 0.68 and the upper CI, 0.95. The specificity and sensitivity of each group (group-specificity&#x2013;sensitivity) were as follows: NAR-0.82&#x2013;0.80, AR-0.86&#x2013;0.80, and FEP-1.00&#x2013;1.00. The results are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. Eight patients in the test set had their statuses at T2. Only one AR patient at T0 was misclassified into NAR by our model, but their status at T2 turned out to be NAR.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Mean absolute SHAP values.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1595197-g002.tif"/>
</fig>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>ROC curve of XGBoostClassifier model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1595197-g003.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Classification report (a), specificity and sensitivity (b), 95% confidence intervals (c).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" colspan="5" align="center">(a) Classification report</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" align="center">Patient group and Metrics</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1-score</th>
<th valign="middle" align="center">Support</th>
</tr>
<tr>
<td valign="middle" align="left">NAR</td>
<td valign="middle" align="right">0.57</td>
<td valign="middle" align="right">0.80</td>
<td valign="middle" align="right">0.67</td>
<td valign="middle" align="right">5</td>
</tr>
<tr>
<td valign="middle" align="left">AR</td>
<td valign="middle" align="right">0.92</td>
<td valign="middle" align="right">0.80</td>
<td valign="middle" align="right">0.86</td>
<td valign="middle" align="right">15</td>
</tr>
<tr>
<td valign="middle" align="left">FEP</td>
<td valign="middle" align="right">1.00</td>
<td valign="middle" align="right">1.00</td>
<td valign="middle" align="right">1.00</td>
<td valign="middle" align="right">2</td>
</tr>
<tr>
<td valign="middle" align="left">Accuracy</td>
<td valign="middle" align="left"/>
<td valign="middle" align="left"/>
<td valign="middle" align="right">0.82</td>
<td valign="middle" align="right"/>
</tr>
<tr>
<td valign="middle" align="left">Macro average</td>
<td valign="middle" align="right">0.83</td>
<td valign="middle" align="right">0.87</td>
<td valign="middle" align="right">0.84</td>
<td valign="middle" align="right"/>
</tr>
<tr>
<td valign="middle" align="left">Weighted average</td>
<td valign="middle" align="right">0.85</td>
<td valign="middle" align="right">0.82</td>
<td valign="middle" align="right">0.83</td>
<td valign="middle" align="left">
</td>
</tr>
</tbody>
</table>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" colspan="5" align="center">(b) Specificity and sensitivity by group</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" align="left">Patient group</th>
<th valign="middle" colspan="2" align="center">Specificity</th>
<th valign="middle" colspan="2" align="center">Sensitivity</th>
</tr>
<tr>
<td valign="middle" align="left">NAR</td>
<td valign="middle" colspan="2" align="right">0.82</td>
<td valign="middle" colspan="2" align="right">0.80</td>
</tr>
<tr>
<td valign="middle" align="left">AR</td>
<td valign="middle" colspan="2" align="right">0.86</td>
<td valign="middle" colspan="2" align="right">0.80</td>
</tr>
<tr>
<td valign="middle" align="left">FEP</td>
<td valign="middle" colspan="2" align="right">1.00</td>
<td valign="middle" colspan="2" align="right">1.00</td>
</tr>
</tbody>
</table>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" colspan="5" align="center">(c) 95% CI</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" align="left">Accuracy and Confidence level</th>
<th valign="middle" colspan="4" align="center">95% Confidence interval (CI)</th>
</tr>
<tr>
<td valign="middle" align="left">Test accuracy</td>
<td valign="middle" colspan="4" align="right">0.82</td>
</tr>
<tr>
<td valign="middle" align="left">Lower CI</td>
<td valign="middle" colspan="4" align="right">0.68</td>
</tr>
<tr>
<td valign="middle" align="left">Upper CI</td>
<td valign="middle" colspan="4" align="right">0.95</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The SHAP values of each individual in each class are visually represented in <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4</bold>
</xref> (NAR), <xref ref-type="fig" rid="f5">
<bold>5</bold>
</xref> (AR), and <xref ref-type="fig" rid="f6">
<bold>6</bold>
</xref> (FEP). The x-axis indicates the SHAP values, the y-axis shows the features, and the color of the point represents the original value of that sample, that is, higher in red and lower in blue. The farther a point is from the center vertical axis, the stronger its impact is on the classification. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> shows that lower scores in intersubjective LSA minimum, lexical density, and subjective LSA without doctor minimum have a negative impact on predictions. In other words, these lower values are indicative of the individuals&#x2019; lower chance of being classified as NAR. Conversely, higher scores, albeit to a lesser degree, in filled pauses and subjective LSA median contribute positively to NAR. The magnitude of the higher scores in the proportion of &#x201c;I&#x201d; to the total number of words suggests their relatively small negative impact on the NAR classification. In <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, the lower proportion of &#x201c;I&#x201d; to the total number of words, and higher frequencies of abandonment/auto-interruption and filled pauses, have a negative impact on predictions in AR. When scores in the proportion of &#x201c;I&#x201d; to the personal pronouns and subjective LSA minimum are higher, the odds on individuals being classified as AR are higher. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> shows that lower minimum scores in all the three types of LSA contribute positively to FEP with the greatest magnitude of intersubjective LSA minimum. Higher values in subjective LSA wodr median negatively impact FEP. The contributions are summarized by patient group, direction, and magnitude in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>SHAP values of Not-At-Risk patients.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1595197-g004.tif"/>
</fig>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>SHAP values of At-Risk patients.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1595197-g005.tif"/>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>SHAP values of First Episode of Psychosis patients.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1595197-g006.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Overview of the directions (positive and negative impacts on classification) and magnitudes (higher and lower values marked with ordinal numbers) of linguistic markers based on SHAP values.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Patient groups</th>
<th valign="middle" colspan="2" align="center">Positive impact on classification</th>
<th valign="middle" colspan="2" align="center">Negative impact on classification</th>
</tr>
<tr>
<th valign="middle" align="center">Higher values</th>
<th valign="middle" align="center">Lower values</th>
<th valign="middle" align="center">Higher values</th>
<th valign="middle" align="center">Lower values</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="3" align="center">NAR</td>
<td valign="middle" align="center">Filled pauses (4th)</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x201c;I&#x201d;/total (5th)</td>
<td valign="middle" align="center">
<italic>Intersubjective LSA minimum</italic> (1st)</td>
</tr>
<tr>
<td valign="middle" align="center">Subjective LSA median (6th)</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Lexical density (2nd)</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Subjective LSA wodr minimum (3rd)</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="center">AR</td>
<td valign="middle" align="center">&#x201c;I&#x201d;/personal pronouns (4th)</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Abandonment/auto-interruption (2nd)</td>
<td valign="middle" align="center">&#x201c;I&#x201d;/total (1st)</td>
</tr>
<tr>
<td valign="middle" align="center">
<italic>Subjective LSA minimum</italic> (5th)</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Filled pauses (3rd)</td>
<td valign="middle" align="center"/>
</tr>
<tr>
<td valign="middle" rowspan="3" align="center">FEP</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">
<italic>Intersubjective LSA minimum</italic> (1st)</td>
<td valign="middle" align="center">Subjective LSA wodr median (4th)</td>
<td valign="middle" align="center"/>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Subjective LSA wodr minimum (2nd)</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">
<italic>Subjective LSA minimum</italic> (3rd)</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Features with SHAP values that are largely spread out across the x-axis, i.e., indicative of both directions, are not included in the table. Feature names in italic = features with significant differences between groups (Kruskal&#x2013;Wallis test).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>The present study aimed at detecting relevant linguistic markers that could classify French-speaking UHR patients by their status at T0 and seeking to explain the classification results with regard to linguistic manifestations observed in the clinic. The results showed that our model based on XGBoost, SMOTE, and the SHAP values could get good performance through the interplay of the four linguistic markers obtained from a feature importance method using the SHAP values on the original data. These mean absolute SHAP values as feature importance revealed that the two uppermost features pertained to semantic coherence, the third most important to the use of &#x201c;I,&#x201d; and the last important feature was one of the disfluency-related elements, filled pauses. The two hypotheses thereby turned out to be true&#x2014;semantic coherence and the use of &#x201c;I&#x201d; played a key role in the classification. The four linguistic markers identified pertain to weak coherence (intersubjective LSA minimum and subjective LSA wodr minimum, i.e., the lowest LSA score in each patient), self-related subject pronoun (the proportion of &#x201c;I&#x201d; to the total number of words), and disfluency (filled pauses).</p>
<p>Semantic incoherence has been reported to be a linguistic characteristic in FEP or schizophrenia (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B20">20</xref>, <xref ref-type="bibr" rid="B40">40</xref>&#x2013;<xref ref-type="bibr" rid="B42">42</xref>, <xref ref-type="bibr" rid="B45">45</xref>, <xref ref-type="bibr" rid="B46">46</xref>, <xref ref-type="bibr" rid="B56">56</xref>). It is noteworthy that lower minimum scores contribute positively to FEP and negatively to NAR regardless of the LSA type. Higher minimum scores in subjective LSA appear to have a positive impact on classifying AR. The feature intersubjective LSA minimum turned out to have significant differences in Kruskal&#x2013;Wallis and Dunn&#x2019;s tests and a much greater impact on predictions than the other markers. This type of coherence was calculated between consecutive pairs of speech turns. Studies on coherence have been focused on patients&#x2019; utterances (<xref ref-type="bibr" rid="B40">40</xref>&#x2013;<xref ref-type="bibr" rid="B44">44</xref>, <xref ref-type="bibr" rid="B49">49</xref>, <xref ref-type="bibr" rid="B52">52</xref>) like subjective LSA wodr (only-patient LSA) in our study. A dialogue is constructed within the framework of turn-taking described as a type of social organization that is implicated in speech exchange systems (<xref ref-type="bibr" rid="B103">103</xref>). For a dialogue to be coherent, a response should be fluent, consistent, context related (<xref ref-type="bibr" rid="B104">104</xref>), and the respondent should understand conventional meaning and catch their interlocutor&#x2019;s intention. Dialogue coherence is thereby grounded in Speech Act Theory (<xref ref-type="bibr" rid="B105">105</xref>, <xref ref-type="bibr" rid="B106">106</xref>) as well as related theories on conversation analysis and discursive pragmatics (<xref ref-type="bibr" rid="B107">107</xref>&#x2013;<xref ref-type="bibr" rid="B109">109</xref>), wherein semantics and pragmatics are entailed. This weak dialogue coherence could partly explain some occasional strange speech and social interaction impairment in patients. Higher median values in subjective LSA contribute positively to NAR classification, whereas higher subjective LSA wodr median scores have a negative impact on FEP. Taken together, these results suggest that weak coherence is a marker of FEP even though it is still somewhat premature to generalize this finding due to the small sample size of FEP in the current study.</p>
<p>The use of the first-person singular pronouns in schizophrenia has been explored in some studies whose results were opposite to one another. When compared to patients with mood disorder, schizophrenics used fewer first-person singular pronouns (<xref ref-type="bibr" rid="B26">26</xref>) whereas these pronouns were more frequent in individuals with schizophrenia than healthy controls (<xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B56">56</xref>). The present study focused on the first-person singular subject pronoun &#x201c;I.&#x201d; The results showed no significant difference between groups, and higher and lower scores of &#x201c;I&#x201d; in FEP do not provide unequivocal contribution types contrary to what has been reported in the literature. However, more frequent use of &#x201c;I&#x201d; has a positive impact on AR classification, whereas it contributes negatively to NAR. The difference between the findings in the aforementioned studies and ours could be due to the differences in the populations compared (mood disorder vs. schizophrenia, healthy individuals vs. people with schizophrenia, NAR vs. FEP, and AR vs. FEP) and the pronouns compared (first-person singular pronouns; first-person singular subjective pronoun). The frequency of &#x201c;I&#x201d; in this study allowed differentiating between NAR and AR. The more frequent use of &#x201c;I&#x201d; in AR might indicate their more intense emotional distress compared to the NAR group as the statuses are the outcome of the CAARMS that assesses &#x201c;emotional disturbance&#x201d; in one of the seven subscales. Rude and colleagues showed that depressed college students used &#x201c;I&#x201d; more frequently&#x2014;not the other first-person singular pronouns such as &#x201c;me&#x201d; or &#x201c;myself&#x201d;&#x2014;than non-depressed peers (<xref ref-type="bibr" rid="B110">110</xref>). The differentiation between NAR and AR by the frequency of &#x201c;I&#x201d; might be indicative of more self-centered speech of AR and explained by their considering the self to be a solitary actor/agent as proposed by Rude and colleagues in (<xref ref-type="bibr" rid="B110">110</xref>). The meaning of higher and lower values in the frequency of &#x201c;I&#x201d; found in both directions in FEP is unclear and intriguing to us, but it might be partly explained by current affective disorders that turned out to be significantly more common in at-risk mental state than FEP (<xref ref-type="bibr" rid="B111">111</xref>). This claim does not refute the interpretation of the aforementioned differentiation between NAR and AR.</p>
<p>A filled pause is an uttered sound that fills a momentary interruption in speech production. When considered a pragmatic function, it has several functions such as discourse planning and structuring, and turn-taking (<xref ref-type="bibr" rid="B112">112</xref>) by signaling delays when a speaker stalls for time to retrieve information and wishes to continue their utterance (<xref ref-type="bibr" rid="B113">113</xref>). When considered a speech disfluency element, filled pauses are symptomatic of production difficulties (<xref ref-type="bibr" rid="B114">114</xref>). In the present study, the feature filled pauses is another marker that allows differentiation between NAR and AR. Its higher values contribute positively to NAR and negatively to AR. No impact of this disfluency element is observed on FEP classification. Another disfluency element, abandonment/auto-interruption, plays a role in classifying AR. When its scores are higher, it has a negative impact on AR predictions. It has been reported that patients with schizophrenia use fewer filled pauses (<xref ref-type="bibr" rid="B74">74</xref>, <xref ref-type="bibr" rid="B115">115</xref>, <xref ref-type="bibr" rid="B116">116</xref>) and produce longer filled pauses than healthy controls (<xref ref-type="bibr" rid="B117">117</xref>). Interestingly, Costa and Silva found that filled pauses before personal pronouns produced by patients with schizophrenia were twice as long as others, and the pronouns are mostly first-person singular pronouns (<xref ref-type="bibr" rid="B117">117</xref>). It was argued by the authors that their result could be explained by patients&#x2019; possible difficulties with self-reference. Filled pauses have ambivalent roles as mentioned above&#x2014;they not only help speech production but also indicate hesitations and difficulties. Lower values in filled pauses in AR in this study, and fewer thereof in FEP in the literature, could be interpreted as indicative of somewhat disturbed pragmatic functions rather than speech disfluency. No contribution of filled pauses to FEP predictions contrary to what has been reported in the literature may be due to different populations compared (schizophrenia vs. FEP) and the small number of FEP patients in the current study.</p>
<p>The present exploratory study used recordings of the first consultations, a non-invasive method that does not transcend the classic healthcare frames, while allowing data collection under the same conditions for all participants. Our results provided evidence that a small number of linguistic markers without demographic or clinical data could classify UHR patients even at T0, that is, when patients do probably not present with obvious abnormalities in language behaviors. Besides, even healthy controls can experience mild language abnormalities (<xref ref-type="bibr" rid="B118">118</xref>), which could make language analyses more subtle and complicated. It should be pointed out that even though the AR patient at T0 who was misclassified into NAR is a single case of the kind in the present study, this misclassification&#x2014;along with the other seven patients with their statuses at T2 who were correctly classified&#x2014;is encouraging. It should cautiously be noted that the small number of FEP along with possible linguistic and cultural differences could make it somewhat delicate to generalize the results. However, the possible linguistic and cultural factor may not intervene in FTD as a systemic review article suggests a three-factor FTD structure with two prominent dimensions (disorganization and negative dimensions) is likely consistent and robust across languages (<xref ref-type="bibr" rid="B119">119</xref>). As a number of studies in the literature have also shown disturbed semantic coherence in FEP and schizophrenia, it could be argued that at least semantic disturbances are a universal linguistic manifestation of patients with psychosis regardless of languages and cultures. The SHAP values provided a local interpretation or the contribution of each feature to the classification. Even some features, such as the frequency of &#x201c;I,&#x201d; filled pauses, subjective LSA wodr minimum, wherein no significant group difference was observed, showed distinctive differences in the directions of the SHAP values and/or the magnitude. These differences would more likely reflect very subtle differences between patient groups recorded at a very early stage of care in psychiatry than an overfitting issue, since the model went through a cross-validation phase, although it was with a small k value. The SHAP explainability method could thereby allow getting deeper insight into the linguistic characteristics and speech patterns of each category of patients, which could lead to improving diagnostic methods.</p>
</sec>
<sec id="s5">
<label>5</label>
<title>Limitation</title>
<p>The current study lacks FEP patients and the 2-year statuses of most patients. In addition, our dataset is relatively small and imbalanced, which led us to carrying out an exploratory study to test the feasibility and potential of a gradient boosting model using only linguistic data. With new transformer-based models, such as BERT and SBERT, as well as word-embedding models, like GloVe, LSA is considered by some to be outdated, despite its advantages, mainly because LSA does not consider word order and context. This weakness might be critical to clinical data. It would therefore be interesting to use a new model combining LSA and BERT (BERT-LSA) (<xref ref-type="bibr" rid="B120">120</xref>) or other models in a future study. The inclusion of more patients and their statuses at T2 would allow more robust models and more accurate model performance evaluations. It is therefore planned to continue to record UHR patients, include more FEP, and analyze their speech using more classifiers for performance comparisons in search of a good diagnostic aid tool.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available due to medical confidentiality. Requests to access the datasets should be directed to D-HK-D, <email xlink:href="mailto:dh.kimdufor@gmail.com">dh.kimdufor@gmail.com</email>.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Comit&#xe9; de Protection des Personnes EST-III (CPP:18.04.03, ID-RCB: 2017-A02702-51). The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation in this study was provided by the participants&#x2019; legal guardians/next of kin.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>DK: Writing &#x2013; review &amp; editing, Conceptualization, Investigation, Writing &#x2013; original draft, Data curation, Formal Analysis, Methodology, Software, Visualization. MW: Conceptualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Funding acquisition, Project administration, Validation. M-OK: Writing &#x2013; review &amp; editing, Funding acquisition, Project administration, Validation. YH: Writing &#x2013; review &amp; editing. PL: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. CL: Writing &#x2013; original draft, Data curation, Validation, Methodology, Investigation, Writing &#x2013; review &amp; editing, Funding acquisition, Conceptualization.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work has been supported by the French government&#x2019;s &#x201c;Investissement d&#x2019;Avenir&#x201d; program, which is managed by the Agence Nationale de la Recherche (ANR), under the reference PsyCARE ANR-18&#x2013;429 RHUS-0014.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We would like to thank the psychiatrists and research nurses at the CEVUP, CHU de Brest, for recording their consultations and helping us out with clinical data. We are also grateful to Catherine and Valentine for the transcription.</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyt.2025.1595197/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpsyt.2025.1595197/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.xlsx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
<supplementary-material xlink:href="Table2.xlsx" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fatouros-Bergman</surname> <given-names>H</given-names>
</name>
<name>
<surname>Cervenka</surname> <given-names>S</given-names>
</name>
<name>
<surname>Flyckt</surname> <given-names>L</given-names>
</name>
<name>
<surname>Edman</surname> <given-names>G</given-names>
</name>
<name>
<surname>Farde</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Meta-analysis of cognitive performance in drug-na&#xef;ve patients with schizophrenia</article-title>. <source>Schizophr Res</source>. (<year>2014</year>) <volume>158</volume>:<page-range>156&#x2013;62</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2014.06.034</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Larsen</surname> <given-names>TK</given-names>
</name>
<name>
<surname>Melle</surname> <given-names>I</given-names>
</name>
<name>
<surname>Auestad</surname> <given-names>B</given-names>
</name>
<name>
<surname>Haahr</surname> <given-names>U</given-names>
</name>
<name>
<surname>Joa</surname> <given-names>I</given-names>
</name>
<name>
<surname>Johannessen</surname> <given-names>JO</given-names>
</name>
<etal/>
</person-group>. <article-title>Early detection of psychosis: positive effects on 5-year outcome</article-title>. <source>psychol Med</source>. (<year>2011</year>) <volume>41</volume>:<page-range>1461&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/S0033291710002023</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murru</surname> <given-names>A</given-names>
</name>
<name>
<surname>Carpiniello</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Duration of untreated illness as a key to early intervention in schizophrenia: a review</article-title>. <source>Neurosci Lett</source>. (<year>2018</year>) <volume>669</volume>:<fpage>59</fpage>&#x2013;<lpage>67</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neulet.2016.10.003</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Krebs</surname> <given-names>MO</given-names>
</name>
</person-group>. <source>Signes pr&#xe9;coces de schizophr&#xe9;nie</source>. <publisher-loc>Paris, France</publisher-loc>: <publisher-name>Dunod</publisher-name> (<year>2015</year>).</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Howes</surname> <given-names>OD</given-names>
</name>
<name>
<surname>Whitehurst</surname> <given-names>T</given-names>
</name>
<name>
<surname>Shatalina</surname> <given-names>E</given-names>
</name>
<name>
<surname>Townsend</surname> <given-names>L</given-names>
</name>
<name>
<surname>Onwordi</surname> <given-names>EC</given-names>
</name>
<name>
<surname>Mak</surname> <given-names>TLA</given-names>
</name>
<etal/>
</person-group>. <article-title>The clinical significance of duration of untreated psychosis: an umbrella review and random-effects meta-analysis</article-title>. <source>World Psychiatry</source>. (<year>2021</year>) <volume>20</volume>:<fpage>75</fpage>&#x2013;<lpage>95</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/wps.20822</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>Correlation between duration of untreated psychosis and long-term prognosis in chronic schizophrenia</article-title>. <source>Front Psychiatry</source>. (<year>2023</year>) <volume>14</volume>:<elocation-id>1112657</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2023.1112657</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fusar-Poli</surname> <given-names>P</given-names>
</name>
<name>
<surname>Cappucciati</surname> <given-names>M</given-names>
</name>
<name>
<surname>Borgwardt</surname> <given-names>S</given-names>
</name>
<name>
<surname>Woods</surname> <given-names>SW</given-names>
</name>
<name>
<surname>Addington</surname> <given-names>J</given-names>
</name>
<name>
<surname>Nelson</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Heterogeneity of psychosis risk within individuals at clinical high risk: a meta-analytical stratification</article-title>. <source>JAMA Psychiatry</source>. (<year>2016</year>) <volume>73</volume>:<page-range>113&#x2013;20</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/jamapsychiatry.2015.2324</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andreasen</surname> <given-names>NC</given-names>
</name>
</person-group>. <article-title>Scale for the assessment of thought, language, and communication (TLC)</article-title>. <source>Schizophr Bull</source>. (<year>1986</year>) <volume>12</volume>:<fpage>473</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/schbul/12.3.473</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Covington</surname> <given-names>MA</given-names>
</name>
<name>
<surname>He</surname> <given-names>C</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>C</given-names>
</name>
<name>
<surname>Na&#xe7;i</surname> <given-names>L</given-names>
</name>
<name>
<surname>McClain</surname> <given-names>JT</given-names>
</name>
<name>
<surname>Fjordbak</surname> <given-names>BS</given-names>
</name>
<etal/>
</person-group>. <article-title>Schizophrenia and the structure of language: the linguist&#x2019;s view</article-title>. <source>Schizophr Res</source>. (<year>2005</year>) <volume>77</volume>:<fpage>85</fpage>&#x2013;<lpage>98</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2005.01.016</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuperberg</surname> <given-names>GR</given-names>
</name>
</person-group>. <article-title>Language in schizophrenia part 1: an introduction</article-title>. <source>Lang Linguistics Compass</source>. (<year>2010</year>) <volume>4</volume>:<page-range>576&#x2013;89</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1749-818X.2010.00216.x</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hinzen</surname> <given-names>W</given-names>
</name>
<name>
<surname>Rossell&#xf3;</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>The linguistics of schizophrenia: thought disturbance as language pathology across positive symptoms</article-title>. <source>Front Psychol</source>. (<year>2015</year>) <volume>6</volume>:<elocation-id>126923</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyg.2015.00971</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ehlen</surname> <given-names>F</given-names>
</name>
<name>
<surname>Montag</surname> <given-names>C</given-names>
</name>
<name>
<surname>Leopold</surname> <given-names>K</given-names>
</name>
<name>
<surname>Heinz</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Linguistic findings in persons with schizophrenia&#x2014;a review of the current literature</article-title>. <source>Front Psychol</source>. (<year>2023</year>) <volume>14</volume>:<elocation-id>1287706</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyg.2023.1287706</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Whitehorn</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Zipf</surname> <given-names>GK</given-names>
</name>
</person-group>. <article-title>Schizophrenic language</article-title>. <source>Arch Neurol Psychiatry</source>. (<year>1943</year>) <volume>49</volume>:<page-range>831&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/archneurpsyc.1943.02290180055006</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lorenz</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Problems posed by schizophrenic language</article-title>. <source>Arch Gen Psychiatry</source>. (<year>1961</year>) <volume>4</volume>:<page-range>603&#x2013;10</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/archpsyc.1961.01710120073008</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wolcott</surname> <given-names>RH</given-names>
</name>
</person-group>. <article-title>Schizophrenese: A private language</article-title>. <source>J Health Soc Behav</source>. (<year>1970</year>) <volume>11</volume>:<page-range>126&#x2013;34</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2307/2948472</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chaika</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>A linguist looks at &#x201c;schizophrenic&#x201d; language</article-title>. <source>Brain Language</source>. (<year>1974</year>) <volume>1</volume>:<page-range>257&#x2013;76</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0093-934X(74)90040-6</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baskak</surname> <given-names>B</given-names>
</name>
<name>
<surname>Ozel</surname> <given-names>ET</given-names>
</name>
<name>
<surname>Atbasoglu</surname> <given-names>EC</given-names>
</name>
<name>
<surname>Baskak</surname> <given-names>SC</given-names>
</name>
</person-group>. <article-title>Peculiar word use as a possible trait marker in schizophrenia</article-title>. <source>Schizophr Res</source>. (<year>2008</year>) <volume>103</volume>:<page-range>311&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2008.04.025</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Champagne-Lavau</surname> <given-names>M</given-names>
</name>
<name>
<surname>Stip</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Pragmatic and executive dysfunction in schizophrenia</article-title>. <source>J Neurolinguistics</source>. (<year>2010</year>) <volume>23</volume>:<page-range>285&#x2013;96</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jneuroling.2009.08.009</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moro</surname> <given-names>A</given-names>
</name>
<name>
<surname>Bambini</surname> <given-names>V</given-names>
</name>
<name>
<surname>Bosia</surname> <given-names>M</given-names>
</name>
<name>
<surname>Anselmetti</surname> <given-names>S</given-names>
</name>
<name>
<surname>Riccaboni</surname> <given-names>R</given-names>
</name>
<name>
<surname>Cappa</surname> <given-names>SF</given-names>
</name>
<etal/>
</person-group>. <article-title>Detecting syntactic and semantic anomalies in schizophrenia</article-title>. <source>Neuropsychologia</source>. (<year>2015</year>) <volume>79</volume>:<page-range>147&#x2013;57</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neuropsychologia.2015.10.030</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>All&#xe9;</surname> <given-names>MC</given-names>
</name>
<name>
<surname>Potheegadoo</surname> <given-names>J</given-names>
</name>
<name>
<surname>K&#xf6;ber</surname> <given-names>C</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>P</given-names>
</name>
<name>
<surname>Coutelle</surname> <given-names>R</given-names>
</name>
<name>
<surname>Habermas</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Impaired coherence of life narratives of patients with schizophrenia</article-title>. <source>Sci Rep</source>. (<year>2015</year>) <volume>5</volume>:<fpage>12934</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/srep12934</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname> <given-names>MH</given-names>
</name>
<name>
<surname>Chance</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Cram</surname> <given-names>DF</given-names>
</name>
<name>
<surname>Crow</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Luc</surname> <given-names>A</given-names>
</name>
<name>
<surname>Hage</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Evidence of pragmatic impairments in speech and proverb interpretation in schizophrenia</article-title>. <source>J Psycholinguist Res</source>. (<year>2015</year>) <volume>44</volume>:<page-range>469&#x2013;83</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10936-014-9298-2</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sass</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Parnas</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Schizophrenia, consciousness, and the self</article-title>. <source>Schizophr Bull</source>. (<year>2003</year>) <volume>29</volume>:<page-range>427&#x2013;44</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/oxfordjournals.schbul.a007017</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nelson</surname> <given-names>B</given-names>
</name>
<name>
<surname>Fornito</surname> <given-names>A</given-names>
</name>
<name>
<surname>Harrison</surname> <given-names>BJ</given-names>
</name>
<name>
<surname>Y&#xfc;cel</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sass</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Yung</surname> <given-names>AR</given-names>
</name>
<etal/>
</person-group>. <article-title>A disturbed sense of self in the psychosis prodrome: Linking phenomenology and neurobiology</article-title>. <source>Neurosci Biobehav Rev</source>. (<year>2009</year>) <volume>33</volume>:<page-range>807&#x2013;17</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neubiorev.2009.01.002</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moe</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Docherty</surname> <given-names>NM</given-names>
</name>
</person-group>. <article-title>Schizophrenia and the sense of self</article-title>. <source>Schizophr Bull</source>. (<year>2014</year>) <volume>40</volume>:<page-range>161&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/schbul/sbt121</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buck</surname> <given-names>B</given-names>
</name>
<name>
<surname>Penn</surname> <given-names>DL</given-names>
</name>
</person-group>. <article-title>Lexical characteristics of emotional narratives in schizophrenia: relationships with symptoms, functioning, and social cognition</article-title>. <source>J Nervous Ment Dis</source>. (<year>2015</year>) <volume>203</volume>:<page-range>702&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/NMD.0000000000000354</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fineberg</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Deutsch-Link</surname> <given-names>S</given-names>
</name>
<name>
<surname>Ichinose</surname> <given-names>M</given-names>
</name>
<name>
<surname>McGuinness</surname> <given-names>T</given-names>
</name>
<name>
<surname>Bessette</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Chung</surname> <given-names>CK</given-names>
</name>
<etal/>
</person-group>. <article-title>Word use in first-person accounts of schizophrenia</article-title>. <source>Br J Psychiatry</source>. (<year>2015</year>) <volume>206</volume>:<page-range>32&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1192/bjp.bp.113.140046</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fineberg</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Leavitt</surname> <given-names>J</given-names>
</name>
<name>
<surname>Deutsch-Link</surname> <given-names>S</given-names>
</name>
<name>
<surname>Dealy</surname> <given-names>S</given-names>
</name>
<name>
<surname>Landry</surname> <given-names>CD</given-names>
</name>
<name>
<surname>Pirruccio</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Self-reference in psychosis and depression: a language marker of illness</article-title>. <source>Psychol Med</source>. (<year>2016</year>) <volume>46</volume>:<page-range>2605&#x2013;15</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/S0033291716001215</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname> <given-names>SX</given-names>
</name>
<name>
<surname>Kriz</surname> <given-names>R</given-names>
</name>
<name>
<surname>Cho</surname> <given-names>S</given-names>
</name>
<name>
<surname>Park</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Harowitz</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gur</surname> <given-names>RE</given-names>
</name>
<etal/>
</person-group>. <article-title>Natural language processing methods are sensitive to sub-clinical linguistic differences in schizophrenia spectrum disorders</article-title>. <source>NPJ Schizophr</source>. (<year>2021</year>) <volume>7</volume>:<fpage>25</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41537-021-00154-3</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chan</surname> <given-names>CC</given-names>
</name>
<name>
<surname>Norel</surname> <given-names>R</given-names>
</name>
<name>
<surname>Agurto</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lysaker</surname> <given-names>PH</given-names>
</name>
<name>
<surname>Myers</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Hazlett</surname> <given-names>EA</given-names>
</name>
<etal/>
</person-group>. <article-title>Emergence of language related to self-experience and agency in autobiographical narratives of individuals with schizophrenia</article-title>. <source>Schizophr Bull</source>. (<year>2023</year>) <volume>49</volume>:<page-range>444&#x2013;53</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/schbul/sbac126</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morice</surname> <given-names>RD</given-names>
</name>
<name>
<surname>Ingram</surname> <given-names>JCL</given-names>
</name>
</person-group>. <article-title>Language analysis in schizophrenia: diagnostic implications</article-title>. <source>Aust N Z J Psychiatry</source>. (<year>1982</year>) <volume>16</volume>:<fpage>11</fpage>&#x2013;<lpage>21</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3109/00048678209161186</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fraser</surname> <given-names>WI</given-names>
</name>
<name>
<surname>King</surname> <given-names>KM</given-names>
</name>
<name>
<surname>Thomas</surname> <given-names>P</given-names>
</name>
<name>
<surname>Kendell</surname> <given-names>RE</given-names>
</name>
</person-group>. <article-title>The diagnosis of schizophrenia by language analysis</article-title>. <source>Br J Psychiatry</source>. (<year>1986</year>) <volume>148</volume>:<page-range>275&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1192/bjp.148.3.275</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thomas</surname> <given-names>P</given-names>
</name>
<name>
<surname>King</surname> <given-names>K</given-names>
</name>
<name>
<surname>Fraser</surname> <given-names>WI</given-names>
</name>
</person-group>. <article-title>Positive and negative symptoms of schizophrenia and linguistic performance</article-title>. <source>Acta Psychiatr Scand</source>. (<year>1987</year>) <volume>76</volume>:<page-range>144&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1600-0447.1987.tb02877.x</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>E</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>CH</given-names>
</name>
<name>
<surname>Lane</surname> <given-names>HY</given-names>
</name>
</person-group>. <article-title>Prediction of functional outcomes of schizophrenia with genetic biomarkers using a bagging ensemble machine learning method with feature selection</article-title>. <source>Sci Rep</source>. (<year>2021</year>) <volume>11</volume>:<fpage>10179</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-021-89540-6</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kraguljac</surname> <given-names>NV</given-names>
</name>
<name>
<surname>McDonald</surname> <given-names>WM</given-names>
</name>
<name>
<surname>Widge</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Rodriguez</surname> <given-names>CI</given-names>
</name>
<name>
<surname>Tohen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Nemeroff</surname> <given-names>CB</given-names>
</name>
</person-group>. <article-title>Neuroimaging biomarkers in schizophrenia</article-title>. <source>Am J Psychiatry</source>. (<year>2021</year>) <volume>178</volume>:<page-range>509&#x2013;21</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1176/appi.ajp.2020.20030340</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodrigues-Amorim</surname> <given-names>D</given-names>
</name>
<name>
<surname>Rivera-Baltan&#xe1;s</surname> <given-names>T</given-names>
</name>
<name>
<surname>L&#xf3;pez</surname> <given-names>M</given-names>
</name>
<name>
<surname>Spuch</surname> <given-names>C</given-names>
</name>
<name>
<surname>Olivares</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Ag&#xed;s-Balboa</surname> <given-names>RC</given-names>
</name>
</person-group>. <article-title>Schizophrenia: a review of potential biomarkers</article-title>. <source>J Psychiatr Res</source>. (<year>2017</year>) <volume>93</volume>:<fpage>37</fpage>&#x2013;<lpage>49</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jpsychires.2017.05.009</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Corcoran</surname> <given-names>CM</given-names>
</name>
<name>
<surname>Cecchi</surname> <given-names>GA</given-names>
</name>
</person-group>. <article-title>Using language processing and speech analysis for the identification of psychosis and other disorders</article-title>. <source>Biol Psychiatry: Cogn Neurosci Neuroimag</source>. (<year>2020</year>) <volume>5</volume>:<page-range>770&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.bpsc.2020.06.004</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Le Glaz</surname> <given-names>A</given-names>
</name>
<name>
<surname>Haralambous</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Kim-Dufor</surname> <given-names>DH</given-names>
</name>
<name>
<surname>Lenca</surname> <given-names>P</given-names>
</name>
<name>
<surname>Billot</surname> <given-names>R</given-names>
</name>
<name>
<surname>Ryan</surname> <given-names>TC</given-names>
</name>
<etal/>
</person-group>. <article-title>Machine learning and natural language processing in mental health: systematic review</article-title>. <source>J Med Internet Res</source>. (<year>2021</year>) <volume>23</volume>:<fpage>e15708</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2196/15708</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Corcoran</surname> <given-names>CM</given-names>
</name>
<name>
<surname>Mittal</surname> <given-names>VA</given-names>
</name>
<name>
<surname>Bearden</surname> <given-names>CE</given-names>
</name>
<name>
<surname>Gur</surname> <given-names>RE</given-names>
</name>
<name>
<surname>Hitczenko</surname> <given-names>K</given-names>
</name>
<name>
<surname>Bilgrami</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>Language as a biomarker for psychosis: a natural language processing approach</article-title>. <source>Schizophr Res</source>. (<year>2020</year>) <volume>226</volume>:<page-range>158&#x2013;66</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2020.04.032</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hitczenko</surname> <given-names>K</given-names>
</name>
<name>
<surname>Mittal</surname> <given-names>VA</given-names>
</name>
<name>
<surname>Goldrick</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Understanding language abnormalities and associated clinical markers in psychosis: the promise of computational methods</article-title>. <source>Schizophr Bull</source>. (<year>2021</year>) <volume>47</volume>:<page-range>344&#x2013;62</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/schbul/sbaa141</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elvev&#xe5;g</surname> <given-names>B</given-names>
</name>
<name>
<surname>Foltz</surname> <given-names>PW</given-names>
</name>
<name>
<surname>Weinberger</surname> <given-names>DR</given-names>
</name>
<name>
<surname>Goldberg</surname> <given-names>TE</given-names>
</name>
</person-group>. <article-title>Quantifying incoherence in speech: an automated methodology and novel application to schizophrenia</article-title>. <source>Schizophr Res</source>. (<year>2007</year>) <volume>93</volume>:<page-range>304&#x2013;16</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2007.03.001</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bedi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Carrillo</surname> <given-names>F</given-names>
</name>
<name>
<surname>Cecchi</surname> <given-names>GA</given-names>
</name>
<name>
<surname>Slezak</surname> <given-names>DF</given-names>
</name>
<name>
<surname>Sigman</surname> <given-names>M</given-names>
</name>
<name>
<surname>Mota</surname> <given-names>NB</given-names>
</name>
<etal/>
</person-group>. <article-title>Automated analysis of free speech predicts psychosis onset in high-risk youths</article-title>. <source>NPJ Schizophr</source>. (<year>2015</year>) <volume>1</volume>:<fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/npjschz.2015.30</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Corcoran</surname> <given-names>CM</given-names>
</name>
<name>
<surname>Carrillo</surname> <given-names>F</given-names>
</name>
<name>
<surname>Fern&#xe1;ndez-Slezak</surname> <given-names>D</given-names>
</name>
<name>
<surname>Bedi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Klim</surname> <given-names>C</given-names>
</name>
<name>
<surname>Javitt</surname> <given-names>DC</given-names>
</name>
<etal/>
</person-group>. <article-title>Prediction of psychosis across protocols and risk cohorts using automated language analysis</article-title>. <source>World Psychiatry</source>. (<year>2018</year>) <volume>17</volume>:<fpage>67</fpage>&#x2013;<lpage>75</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/wps.20491</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morgan</surname> <given-names>SE</given-names>
</name>
<name>
<surname>Diederen</surname> <given-names>K</given-names>
</name>
<name>
<surname>V&#xe9;rtes</surname> <given-names>PE</given-names>
</name>
<name>
<surname>Ip</surname> <given-names>SHY</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>B</given-names>
</name>
<name>
<surname>Thompson</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Natural Language Processing markers in first episode psychosis and people at clinical high-risk</article-title>. <source>Transl Psychiatry</source>. (<year>2021</year>) <volume>11</volume>:<fpage>630</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41398-021-01722-y</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gargano</surname> <given-names>G</given-names>
</name>
<name>
<surname>Caletti</surname> <given-names>E</given-names>
</name>
<name>
<surname>Perlini</surname> <given-names>C</given-names>
</name>
<name>
<surname>Turtulici</surname> <given-names>N</given-names>
</name>
<name>
<surname>Bellani</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bonivento</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Language production impairments in patients with a first episode of psychosis</article-title>. <source>PloS One</source>. (<year>2022</year>) <volume>17</volume>:<fpage>e0272873</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0272873</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mota</surname> <given-names>NB</given-names>
</name>
<name>
<surname>Vasconcelos</surname> <given-names>NAP</given-names>
</name>
<name>
<surname>Lemos</surname> <given-names>N</given-names>
</name>
<name>
<surname>Pieretti</surname> <given-names>AC</given-names>
</name>
<name>
<surname>Kinouchi</surname> <given-names>O</given-names>
</name>
<name>
<surname>Cecchi</surname> <given-names>GA</given-names>
</name>
<etal/>
</person-group>. <article-title>Speech graphs provide a quantitative measure of thought disorder in psychosis</article-title>. <source>PloS One</source>. (<year>2012</year>) <volume>7</volume>:<fpage>e34928</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0034928</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mota</surname> <given-names>NB</given-names>
</name>
<name>
<surname>Furtado</surname> <given-names>R</given-names>
</name>
<name>
<surname>Maia</surname> <given-names>PP</given-names>
</name>
<name>
<surname>Copelli</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ribeiro</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Graph analysis of dream reports is especially informative about psychosis</article-title>. <source>Sci Rep</source>. (<year>2014</year>) <volume>4</volume>:<fpage>3691</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/srep03691</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mota</surname> <given-names>NB</given-names>
</name>
<name>
<surname>Copelli</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ribeiro</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Thought disorder measured as random speech structure classifies negative symptoms and schizophrenia diagnosis 6 months in advance</article-title>. <source>NPJ Schizophr</source>. (<year>2017</year>) <volume>3</volume>:<fpage>18</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41537-017-0019-3</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spencer</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Thompson</surname> <given-names>B</given-names>
</name>
<name>
<surname>Oliver</surname> <given-names>D</given-names>
</name>
<name>
<surname>Diederen</surname> <given-names>K</given-names>
</name>
<name>
<surname>Demjaha</surname> <given-names>A</given-names>
</name>
<name>
<surname>Weinstein</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Lower speech connectedness linked to incidence of psychosis in people at clinical high risk</article-title>. <source>Schizophr Res</source>. (<year>2021</year>) <volume>228</volume>:<fpage>493</fpage>&#x2013;<lpage>501</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2020.09.002</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Figueroa-Barra</surname> <given-names>A</given-names>
</name>
<name>
<surname>Del Aguila</surname> <given-names>D</given-names>
</name>
<name>
<surname>Cerda</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gaspar</surname> <given-names>PA</given-names>
</name>
<name>
<surname>Terissi</surname> <given-names>LD</given-names>
</name>
<name>
<surname>Dur&#xe1;n</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Automatic language analysis identifies and predicts schizophrenia in first-episode of psychosis</article-title>. <source>Schizophrenia</source>. (<year>2022</year>) <volume>8</volume>:<fpage>53</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41537-022-00259-3</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oomen</surname> <given-names>PP</given-names>
</name>
<name>
<surname>De Boer</surname> <given-names>JN</given-names>
</name>
<name>
<surname>Brederoo</surname> <given-names>SG</given-names>
</name>
<name>
<surname>Voppel</surname> <given-names>AE</given-names>
</name>
<name>
<surname>Brand</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Wijnen</surname> <given-names>FNK</given-names>
</name>
<etal/>
</person-group>. <article-title>Characterizing speech heterogeneity in schizophrenia-spectrum disorders</article-title>. <source>J Psychopathol Clin Sci</source>. (<year>2022</year>) <volume>131</volume>:<page-range>172&#x2013;81</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1037/abn0000736</pub-id>
</citation>
</ref>
<ref id="B51">
<label>51</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bambini</surname> <given-names>V</given-names>
</name>
<name>
<surname>Frau</surname> <given-names>F</given-names>
</name>
<name>
<surname>Bischetti</surname> <given-names>L</given-names>
</name>
<name>
<surname>Cuoco</surname> <given-names>F</given-names>
</name>
<name>
<surname>Bechi</surname> <given-names>M</given-names>
</name>
<name>
<surname>Buonocore</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Deconstructing heterogeneity in schizophrenia through language: a semi-automated linguistic analysis and data-driven clustering approach</article-title>. <source>Schizophr</source>. (<year>2022</year>) <volume>8</volume>:<fpage>102</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41537-022-00306-z</pub-id>
</citation>
</ref>
<ref id="B52">
<label>52</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Voppel</surname> <given-names>AE</given-names>
</name>
<name>
<surname>De Boer</surname> <given-names>JN</given-names>
</name>
<name>
<surname>Brederoo</surname> <given-names>SG</given-names>
</name>
<name>
<surname>Schnack</surname> <given-names>HG</given-names>
</name>
<name>
<surname>Sommer</surname> <given-names>IEC</given-names>
</name>
</person-group>. <article-title>Semantic and acoustic markers in schizophrenia-spectrum disorders: A combinatory machine learning approach</article-title>. <source>Schizophr Bull</source>. (<year>2023</year>) <volume>49</volume>:<page-range>S163&#x2013;71</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/schbul/sbac142</pub-id>
</citation>
</ref>
<ref id="B53">
<label>53</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Foster</surname> <given-names>P</given-names>
</name>
<name>
<surname>Tonkyn</surname> <given-names>A</given-names>
</name>
<name>
<surname>Wigglesworth</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Measuring spoken language: A unit for all reasons</article-title>. <source>Appl Linguist</source>. (<year>2000</year>) <volume>21</volume>:<page-range>354&#x2013;75</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/applin/21.3.354</pub-id>
</citation>
</ref>
<ref id="B54">
<label>54</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Templin</surname> <given-names>M</given-names>
</name>
</person-group>. <source>Certain language skills in children: their development and interrelationships</source>. <publisher-loc>Minneapolis</publisher-loc>: <publisher-name>University of Minnesota Press</publisher-name> (<year>1957</year>).</citation>
</ref>
<ref id="B55">
<label>55</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ure</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Lexical density and register differentiation</article-title>. <source>Appl Linguist</source>. (<year>1971</year>) <volume>23</volume>:<page-range>443&#x2013;52</page-range>.</citation>
</ref>
<ref id="B56">
<label>56</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundin</surname> <given-names>NB</given-names>
</name>
<name>
<surname>Cowan</surname> <given-names>HR</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>DK</given-names>
</name>
<name>
<surname>Moe</surname> <given-names>AM</given-names>
</name>
</person-group>. <article-title>Lower cohesion and altered first-person pronoun usage in the spoken life narratives of individuals with schizophrenia</article-title>. <source>Schizophr Res</source>. (<year>2023</year>) <volume>259</volume>:<page-range>140&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2023.04.001</pub-id>
</citation>
</ref>
<ref id="B57">
<label>57</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pointal</surname> <given-names>L</given-names>
</name>
</person-group>. <source>TreeTaggerWrapper. Laboratoire d&#x2019;Informatique pour la M&#xe9;canique et les Sciences de l&#x2019;Ing&#xe9;nieur, Laboratoire Interdisciplinaire des Sciences du Num&#xe9;rique</source>. <publisher-loc>Paris, France</publisher-loc>: <publisher-name>CNRS</publisher-name> (<year>2016</year>).</citation>
</ref>
<ref id="B58">
<label>58</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Halliday</surname> <given-names>MAK</given-names>
</name>
<name>
<surname>Matthiessen</surname> <given-names>CM</given-names>
</name>
</person-group>. <source>Halliday&#x2019;s introduction to functional grammar</source>. <publisher-loc>Milton Park, Abingdon, UK</publisher-loc>: <publisher-name>Routledge</publisher-name> (<year>2013</year>).</citation>
</ref>
<ref id="B59">
<label>59</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cook</surname> <given-names>WA</given-names>
</name>
</person-group>. <source>Introduction to tagmemic analysis</source>. <publisher-loc>Washington D.C., USA</publisher-loc>: <publisher-name>Georgetown University Press</publisher-name> (<year>1969</year>).</citation>
</ref>
<ref id="B60">
<label>60</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bever</surname> <given-names>TG</given-names>
</name>
<name>
<surname>Lackner</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kirk</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>The underlying structures of sentences are the primary units of immediate speech processing</article-title>. <source>Percept Psychophys</source>. (<year>1969</year>) <volume>5</volume>:<page-range>225&#x2013;34</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3758/BF03210545</pub-id>
</citation>
</ref>
<ref id="B61">
<label>61</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Szmrecs&#xe1;nyi</surname> <given-names>B</given-names>
</name>
</person-group>. <source>On operationalizing syntactic complexity</source>. <publisher-loc>Le poids des mots. Proceedings of the 7th international conference on textual data statistical analysis. Louvain-la-Neuve.</publisher-loc> (<year>2004</year>). <volume>2</volume>:<page-range>1032&#x2013;9</page-range>.</citation>
</ref>
<ref id="B62">
<label>62</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Landauer</surname> <given-names>TK</given-names>
</name>
<name>
<surname>Foltz</surname> <given-names>PW</given-names>
</name>
<name>
<surname>Laham</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>An introduction to latent semantic analysis</article-title>. <source>Discourse Processes</source>. (<year>1998</year>) <volume>25</volume>:<page-range>259&#x2013;84</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/01638539809545028</pub-id>
</citation>
</ref>
<ref id="B63">
<label>63</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Landauer</surname> <given-names>TK</given-names>
</name>
<name>
<surname>McNamara</surname> <given-names>DS</given-names>
</name>
<name>
<surname>Dennis</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kintsch</surname> <given-names>W</given-names>
</name>
</person-group>. <source>Handbook of latent semantic analysis</source>. <publisher-loc>Milton Park, Abingdon, UK</publisher-loc>: <publisher-name>Routledge</publisher-name> (<year>2011</year>).</citation>
</ref>
<ref id="B64">
<label>64</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elvev&#xe5;g</surname> <given-names>B</given-names>
</name>
<name>
<surname>Foltz</surname> <given-names>PW</given-names>
</name>
<name>
<surname>Rosenstein</surname> <given-names>M</given-names>
</name>
<name>
<surname>DeLisi</surname> <given-names>LE</given-names>
</name>
</person-group>. <article-title>An automated method to analyze language use in patients with schizophrenia and their first-degree relatives</article-title>. <source>J Neurolinguistics</source>. (<year>2010</year>) <volume>23</volume>:<page-range>270&#x2013;84</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jneuroling.2009.05.002</pub-id>
</citation>
</ref>
<ref id="B65">
<label>65</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holshausen</surname> <given-names>K</given-names>
</name>
<name>
<surname>Harvey</surname> <given-names>PD</given-names>
</name>
<name>
<surname>Elvev&#xe5;g</surname> <given-names>B</given-names>
</name>
<name>
<surname>Foltz</surname> <given-names>PW</given-names>
</name>
<name>
<surname>Bowie</surname> <given-names>CR</given-names>
</name>
</person-group>. <article-title>Latent semantic variables are associated with formal thought disorder and adaptive behavior in older inpatients with schizophrenia</article-title>. <source>Cortex</source>. (<year>2014</year>) <volume>55</volume>:<fpage>88</fpage>&#x2013;<lpage>96</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cortex.2013.02.006</pub-id>
</citation>
</ref>
<ref id="B66">
<label>66</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wiemer-Hastings</surname> <given-names>P</given-names>
</name>
</person-group>. <source>How latent is latent semantic analysis</source>? Proceedings of the 16th international joint conference on Artificial intelligence. <publisher-loc>San Francisco, CA</publisher-loc> (<year>1999</year>) p. <page-range>932&#x2013;7</page-range>.</citation>
</ref>
<ref id="B67">
<label>67</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Graesser</surname> <given-names>AC</given-names>
</name>
<name>
<surname>McNamara</surname> <given-names>DS</given-names>
</name>
<name>
<surname>Louwerse</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>Coh-Metrix: Analysis of text on cohesion and language</article-title>. <source>Behav Res Methods Instrum Computers</source>. (<year>2004</year>) <volume>36</volume>:<fpage>193</fpage>&#x2013;<lpage>202</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3758/BF03195564</pub-id>
</citation>
</ref>
<ref id="B68">
<label>68</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>McNamara</surname> <given-names>DS</given-names>
</name>
<name>
<surname>Graesser</surname> <given-names>AC</given-names>
</name>
<name>
<surname>McCarthy</surname> <given-names>PM</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Z</given-names>
</name>
</person-group>. <source>Automated Evaluation of Text and Discourse with Coh-Metrix</source>. <edition>1st ed</edition>. <publisher-loc>Cambridge, UK</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name> (<year>2014</year>). Available at: <uri xlink:href="https://www.cambridge.org/core/product/identifier/9780511894664/type/book">https://www.cambridge.org/core/product/identifier/9780511894664/type/book</uri> (Accessed <access-date>January 20, 2025</access-date>).</citation>
</ref>
<ref id="B69">
<label>69</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Willits</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Rubin</surname> <given-names>T</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>MN</given-names>
</name>
<name>
<surname>Minor</surname> <given-names>KS</given-names>
</name>
<name>
<surname>Lysaker</surname> <given-names>PH</given-names>
</name>
</person-group>. <article-title>Evidence of disturbances of deep levels of semantic cohesion within personal narratives in schizophrenia</article-title>. <source>Schizophr Res</source>. (<year>2018</year>) <volume>197</volume>:<page-range>365&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2017.11.014</pub-id>
</citation>
</ref>
<ref id="B70">
<label>70</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname> <given-names>T</given-names>
</name>
<name>
<surname>Hespos</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Horton</surname> <given-names>WS</given-names>
</name>
<name>
<surname>Mittal</surname> <given-names>VA</given-names>
</name>
</person-group>. <article-title>Automated analysis of written narratives reveals abnormalities in referential cohesion in youth at ultra high risk for psychosis</article-title>. <source>Schizophr Res</source>. (<year>2018</year>) <volume>192</volume>:<page-range>82&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2017.04.025</pub-id>
</citation>
</ref>
<ref id="B71">
<label>71</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mackinley</surname> <given-names>M</given-names>
</name>
<name>
<surname>Chan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ke</surname> <given-names>H</given-names>
</name>
<name>
<surname>Dempster</surname> <given-names>K</given-names>
</name>
<name>
<surname>Palaniyappan</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Linguistic determinants of formal thought disorder in first episode psychosis</article-title>. <source>Early Intervent Psych</source>. (<year>2021</year>) <volume>15</volume>:<page-range>344&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/eip.12948</pub-id>
</citation>
</ref>
<ref id="B72">
<label>72</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lemaire</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Limites de la lemmatisation pour l&#x2019;extraction de significations</article-title>. <source>In</source>. (<year>2008</year>) <volume>p</volume>:<page-range>725&#x2013;32</page-range>.</citation>
</ref>
<ref id="B73">
<label>73</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Harley</surname> <given-names>TA</given-names>
</name>
</person-group>. <source>The psychology of language: From data to theory</source>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>Psychology press</publisher-name> (<year>2013</year>).</citation>
</ref>
<ref id="B74">
<label>74</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Howes</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lavelle</surname> <given-names>M</given-names>
</name>
<name>
<surname>Healey</surname> <given-names>PG</given-names>
</name>
<name>
<surname>Hough</surname> <given-names>J</given-names>
</name>
<name>
<surname>McCabe</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Disfluencies in dialogues with patients with schizophrenia</article-title>. <source>Proceedings of the Annual Meeting of the Cognitive Science Society</source> (<year>2017</year>) <volume>39</volume>.</citation>
</ref>
<ref id="B75">
<label>75</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vail</surname> <given-names>AK</given-names>
</name>
<name>
<surname>Liebson</surname> <given-names>E</given-names>
</name>
<name>
<surname>Baker</surname> <given-names>JT</given-names>
</name>
<name>
<surname>Morency</surname> <given-names>LP</given-names>
</name>
</person-group>. <article-title>Toward objective, multifaceted characterization of psychotic disorders: Lexical, structural, and disfluency markers of spoken language</article-title>. <source>Proceedings of the 20th ACM International Conference on Multimodal Interaction</source> (<year>2018</year>), <page-range>170&#x2013;178</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3242969</pub-id>
</citation>
</ref>
<ref id="B76">
<label>76</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xc7;okal</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zimmerer</surname> <given-names>V</given-names>
</name>
<name>
<surname>Turkington</surname> <given-names>D</given-names>
</name>
<name>
<surname>Ferrier</surname> <given-names>N</given-names>
</name>
<name>
<surname>Varley</surname> <given-names>R</given-names>
</name>
<name>
<surname>Watson</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Disturbing the rhythm of thought: Speech pausing patterns in schizophrenia, with and without formal thought disorder</article-title>. <source>PloS One</source>. (<year>2019</year>) <volume>14</volume>:<fpage>e0217404</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0217404</pub-id>
</citation>
</ref>
<ref id="B77">
<label>77</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Virtanen</surname> <given-names>P</given-names>
</name>
<name>
<surname>Gommers</surname> <given-names>R</given-names>
</name>
<name>
<surname>Oliphant</surname> <given-names>TE</given-names>
</name>
<name>
<surname>Haberland</surname> <given-names>M</given-names>
</name>
<name>
<surname>Reddy</surname> <given-names>T</given-names>
</name>
<name>
<surname>Cournapeau</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>SciPy 1.0: fundamental algorithms for scientific computing in Python</article-title>. <source>Nat Methods</source>. (<year>2020</year>) <volume>17</volume>:<page-range>352&#x2013;2</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41592-020-0772-5</pub-id>
</citation>
</ref>
<ref id="B78">
<label>78</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Seabold</surname> <given-names>S</given-names>
</name>
<name>
<surname>Perktold</surname> <given-names>J</given-names>
</name>
</person-group>. <source>Statsmodels: Econometric and Statistical Modeling with Python</source>. <publisher-loc>Austin, Texas</publisher-loc>: <publisher-name>SciPy</publisher-name> (<year>2010</year>) p. <page-range>92&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.25080/Majora-92bf1922-011</pub-id>
</citation>
</ref>
<ref id="B79">
<label>79</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>XGBoost: A scalable tree boosting system</article-title>. In: <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. <publisher-name>ACM</publisher-name>, <publisher-loc>San Francisco California USA</publisher-loc> (<year>2016</year>). p. <page-range>785&#x2013;94</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id>
</citation>
</ref>
<ref id="B80">
<label>80</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname> <given-names>JH</given-names>
</name>
</person-group>. <article-title>Greedy function approximation: a gradient boosting machine</article-title>. <source>Ann Stat</source>. (<year>2001</year>) <volume>29</volume>:<page-range>1189&#x2013;232</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1214/aos/1013203451</pub-id>
</citation>
</ref>
<ref id="B81">
<label>81</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>He</surname> <given-names>T</given-names>
</name>
<name>
<surname>Benesty</surname> <given-names>M</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Understand your dataset with XGBoost</article-title>. Available online at: <uri xlink:href="https://cran.r-project.org/web/packages/xgboost/vignettes/discoverYourData.html">https://cran.r-project.org/web/packages/xgboost/vignettes/discoverYourData.html</uri> (Accessed <access-date>March 11, 2024</access-date>).</citation>
</ref>
<ref id="B82">
<label>82</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname> <given-names>M</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>WG</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>QH</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>YC</given-names>
</name>
<name>
<surname>Li</surname> <given-names>ML</given-names>
</name>
</person-group>. <article-title>Optimized XGBoost model with small dataset for predicting relative density of Ti-6Al-4V parts manufactured by selective laser melting</article-title>. <source>Materials</source>. (<year>2022</year>) <volume>15</volume>:<fpage>5298</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ma15155298</pub-id>
</citation>
</ref>
<ref id="B83">
<label>83</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>P</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>X</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>W</given-names>
</name>
</person-group>. <article-title>Small data machine learning in materials science</article-title>. <source>NPJ Comput Mater</source>. (<year>2023</year>) <volume>9</volume>:<fpage>42</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41524-023-01000-z</pub-id>
</citation>
</ref>
<ref id="B84">
<label>84</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chawla</surname> <given-names>NV</given-names>
</name>
<name>
<surname>Bowyer</surname> <given-names>KW</given-names>
</name>
<name>
<surname>Hall</surname> <given-names>LO</given-names>
</name>
<name>
<surname>Kegelmeyer</surname> <given-names>WP</given-names>
</name>
</person-group>. <article-title>SMOTE: synthetic minority over-sampling technique</article-title>. <source>jair</source>. (<year>2002</year>) <volume>16</volume>:<page-range>321&#x2013;57</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1613/jair.953</pub-id>
</citation>
</ref>
<ref id="B85">
<label>85</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diz</surname> <given-names>J</given-names>
</name>
<name>
<surname>Marreiros</surname> <given-names>G</given-names>
</name>
<name>
<surname>Freitas</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Applying data mining techniques to improve breast cancer diagnosis</article-title>. <source>J Med Syst</source>. (<year>2016</year>) <volume>40</volume>:<fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10916-016-0561-y</pub-id>
</citation>
</ref>
<ref id="B86">
<label>86</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramezankhani</surname> <given-names>A</given-names>
</name>
<name>
<surname>Pournik</surname> <given-names>O</given-names>
</name>
<name>
<surname>Shahrabi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Azizi</surname> <given-names>F</given-names>
</name>
<name>
<surname>Hadaegh</surname> <given-names>F</given-names>
</name>
<name>
<surname>Khalili</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>The impact of oversampling with SMOTE on the performance of 3 classifiers in prediction of type 2 diabetes</article-title>. <source>Med Decis Making</source>. (<year>2016</year>) <volume>36</volume>:<page-range>137&#x2013;44</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/0272989X14560647</pub-id>
</citation>
</ref>
<ref id="B87">
<label>87</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdoh</surname> <given-names>SF</given-names>
</name>
<name>
<surname>Abo Rizka</surname> <given-names>M</given-names>
</name>
<name>
<surname>Maghraby</surname> <given-names>FA</given-names>
</name>
</person-group>. <article-title>Cervical cancer diagnosis using random forest classifier with SMOTE and feature reduction techniques</article-title>. <source>IEEE Access</source>. (<year>2018</year>) <volume>6</volume>:<page-range>59475&#x2013;85</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2018.2874063</pub-id>
</citation>
</ref>
<ref id="B88">
<label>88</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fotouhi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Asadi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kattan</surname> <given-names>MW</given-names>
</name>
</person-group>. <article-title>A comprehensive data level analysis for cancer diagnosis on imbalanced data</article-title>. <source>J Biomed Inform</source>. (<year>2019</year>) <volume>90</volume>:<fpage>103089</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jbi.2018.12.003</pub-id>
</citation>
</ref>
<ref id="B89">
<label>89</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Polat</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>A hybrid approach to Parkinson disease classification using speech signal: the combination of SMOTE and random forests</article-title>. In: <source>2019 Scientific Meeting on Electrical-Electronics &amp; Biomedical Engineering and Computer Science (EBBT)</source>. <publisher-name>IEEE</publisher-name>, <publisher-loc>Istanbul, Turkey</publisher-loc> (<year>2019</year>). p. <fpage>1</fpage>&#x2013;<lpage>3</lpage>. Available at: <uri xlink:href="https://ieeexplore.ieee.org/document/8741725/">https://ieeexplore.ieee.org/document/8741725/</uri> (Accessed <access-date>July 21, 2024</access-date>).</citation>
</ref>
<ref id="B90">
<label>90</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shuja</surname> <given-names>M</given-names>
</name>
<name>
<surname>Mittal</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zaman</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Effective prediction of type II diabetes mellitus using data mining classifiers and SMOTE</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Sharma</surname> <given-names>H</given-names>
</name>
<name>
<surname>Govindan</surname> <given-names>K</given-names>
</name>
<name>
<surname>Poonia</surname> <given-names>RC</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>S</given-names>
</name>
<name>
<surname>El-Medany</surname> <given-names>WM</given-names>
</name>
</person-group>, editors. <source>Advances in Computing and Intelligent Systems</source>. <publisher-name>Springer Singapore</publisher-name>, <publisher-loc>Singapore</publisher-loc> (<year>2020</year>). p. <fpage>195</fpage>&#x2013;<lpage>211</lpage>. Algorithms for Intelligent Systems. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-981-15-0222-4_17</pub-id>
</citation>
</ref>
<ref id="B91">
<label>91</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Abdu-Aljabar</surname> <given-names>RD</given-names>
</name>
<name>
<surname>Awad</surname> <given-names>OA</given-names>
</name>
</person-group>. <source>A comparative analysis study of lung cancer detection and relapse prediction using XGBoost classifier</source> Vol. <volume>p</volume>. <publisher-loc>Bristol, UK</publisher-loc>: <publisher-name>IOP Publishing</publisher-name> (<year>2021</year>). p. <fpage>012048</fpage>.</citation>
</ref>
<ref id="B92">
<label>92</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shen</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gan</surname> <given-names>D</given-names>
</name>
<name>
<surname>An</surname> <given-names>B</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>A hybrid method to predict postoperative survival of lung cancer using improved SMOTE and adaptive SVM</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Cesarelli</surname> <given-names>M</given-names>
</name>
</person-group>, editor. <source>Computational and Mathematical Methods in Medicine</source>, vol. <volume>2021</volume> (<year>2021</year>) <publisher-loc>USA</publisher-loc>: <publisher-name>Wiley Online Library</publisher-name>. p. <fpage>1</fpage>&#x2013;<lpage>15</lpage>.</citation>
</ref>
<ref id="B93">
<label>93</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname> <given-names>CC</given-names>
</name>
<name>
<surname>Li</surname> <given-names>YZ</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>HC</given-names>
</name>
<name>
<surname>Tseng</surname> <given-names>MH</given-names>
</name>
</person-group>. <article-title>Melanoma detection using XGB classifier combined with feature extraction and K-means SMOTE techniques</article-title>. <source>Diagnostics</source>. (<year>2022</year>) <volume>12</volume>:<fpage>1747</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics12071747</pub-id>
</citation>
</ref>
<ref id="B94">
<label>94</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karamti</surname> <given-names>H</given-names>
</name>
<name>
<surname>Alharthi</surname> <given-names>R</given-names>
</name>
<name>
<surname>Anizi</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Alhebshi</surname> <given-names>RM</given-names>
</name>
<name>
<surname>Eshmawi</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Alsubai</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Improving prediction of cervical cancer using KNN imputed SMOTE features and multi-model ensemble learning approach</article-title>. <source>Cancers</source>. (<year>2023</year>) <volume>15</volume>:<fpage>4412</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cancers15174412</pub-id>
</citation>
</ref>
<ref id="B95">
<label>95</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhavani</surname> <given-names>CH</given-names>
</name>
<name>
<surname>Govardhan</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Cervical cancer prediction using stacked ensemble algorithm with SMOTE and RFERF</article-title>. <source>Mater Today: Proc</source>. (<year>2023</year>) <volume>80</volume>:<page-range>3451&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.matpr.2021.07.269</pub-id>
</citation>
</ref>
<ref id="B96">
<label>96</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srinivasan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Ramadass</surname> <given-names>P</given-names>
</name>
<name>
<surname>Mathivanan</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Panneer Selvam</surname> <given-names>K</given-names>
</name>
<name>
<surname>Shivahare</surname> <given-names>BD</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>MA</given-names>
</name>
</person-group>. <article-title>Detection of Parkinson disease using multiclass machine learning approach</article-title>. <source>Sci Rep</source>. (<year>2024</year>) <volume>14</volume>:<fpage>13813</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-024-64004-9</pub-id>
</citation>
</ref>
<ref id="B97">
<label>97</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Herawati</surname> <given-names>BC</given-names>
</name>
<name>
<surname>Hairani</surname> <given-names>H</given-names>
</name>
<name>
<surname>Guterres</surname> <given-names>JX</given-names>
</name>
</person-group>. <article-title>SMOTE variants and random forest method: A comprehensive approach to breast cancer classification</article-title>. <source>IJEC</source>. (<year>2024</year>) <volume>3</volume>:<fpage>12</fpage>&#x2013;<lpage>23</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.58291/ijec.v3i1.147</pub-id>
</citation>
</ref>
<ref id="B98">
<label>98</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martinez-Cantin</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>BayesOpt: A Bayesian optimization library for nonlinear optimization, experimental design and bandits</article-title>. <source>J Machine Learning Res</source>. (<year>2014</year>) <volume>15</volume>:<page-range>3915&#x2013;9</page-range>. Available online at: <uri xlink:href="https://arxiv.org/abs/1405.7430">https://arxiv.org/abs/1405.7430</uri>.</citation>
</ref>
<ref id="B99">
<label>99</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>SI</given-names>
</name>
</person-group>. <source>A unified approach to interpreting model predictions. Advances in neural information processing systems</source>, Vol. <volume>30</volume>. <publisher-loc>San Diego, USA</publisher-loc>: <publisher-name>Neural Information Processing Systems Foundation, Inc.</publisher-name> (<year>2017</year>).</citation>
</ref>
<ref id="B100">
<label>100</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>An introduction to explainable AI with Shapley values</article-title>(<year>2018</year>). Available online at: <uri xlink:href="https://shap.readthedocs.io/en/latest/example_notebooks/overviews/An%20introduction%20to%20explainable%20AI%20with%20Shapley%20values.html">https://shap.readthedocs.io/en/latest/example_notebooks/overviews/An%20introduction%20to%20explainable%20AI%20with%20Shapley%20values.html</uri> (Accessed <access-date>April 23, 2024</access-date>).</citation>
</ref>
<ref id="B101">
<label>101</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Hancock</surname> <given-names>JT</given-names>
</name>
<name>
<surname>Khoshgoftaar</surname> <given-names>TM</given-names>
</name>
</person-group>. <article-title>Feature selection strategies: a comparative analysis of SHAP-value and importance-based methods</article-title>. <source>J Big Data</source>. (<year>2024</year>) <volume>11</volume>:<fpage>44</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s40537-024-00905-w</pub-id>
</citation>
</ref>
<ref id="B102">
<label>102</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shapley</surname> <given-names>LS</given-names>
</name>
</person-group>. <source>Notes on the N-Person Game &#x2014; II: The Value of an N-Person Game</source>. <publisher-loc>Santa Monica, USA</publisher-loc>: <publisher-name>Rand Corporation</publisher-name> (<year>1951</year>). Available at: <uri xlink:href="https://www.rand.org/content/dam/rand/pubs/research_memoranda/2008/RM670.pdf">https://www.rand.org/content/dam/rand/pubs/research_memoranda/2008/RM670.pdf</uri> (Accessed <access-date>May 7, 2023</access-date>).</citation>
</ref>
<ref id="B103">
<label>103</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Sacks</surname> <given-names>H</given-names>
</name>
<name>
<surname>Schegloff</surname> <given-names>EA</given-names>
</name>
<name>
<surname>Jefferson</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>A Simplest Systematics for the Organization of Turn Taking for Conversation**This chapter is a variant version of &#x201c;A Simplest Systematics for the Organization of Turn-Taking for Conversation,&#x201d; which was printed in Language, 50, 4</article-title>. Available online at: <uri xlink:href="https://linkinghub.elsevier.com/retrieve/pii/B9780126235500500082">https://linkinghub.elsevier.com/retrieve/pii/B9780126235500500082</uri> (Accessed <access-date>November 12, 2024</access-date>).</citation>
</ref>
<ref id="B104">
<label>104</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>L</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Towards quantifiable dialogue coherence evaluation</article-title>. <source>arXiv preprint arXiv:2106.00507</source>. (<year>2021</year>). Available online at: <uri xlink:href="http://arxiv.org/abs/2106.00507">http://arxiv.org/abs/2106.00507</uri>.</citation>
</ref>
<ref id="B105">
<label>105</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Austin</surname> <given-names>JL</given-names>
</name>
</person-group>. <source>w to do things with words: the William James lectures delivered at Harvard University in 1955</source>. <person-group person-group-type="editor">
<name>
<surname>Urmson</surname> <given-names>JO</given-names>
</name>
</person-group>, editor. <publisher-loc>London</publisher-loc>: <publisher-name>Oxford Univ. Press</publisher-name> (<year>1971</year>). <fpage>166 p</fpage>. Oxford paperbacks.</citation>
</ref>
<ref id="B106">
<label>106</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Searle</surname> <given-names>JR</given-names>
</name>
</person-group>. <source>Speech acts: an essay in the philosophy of language</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge Univ. Press</publisher-name> (<year>1970</year>). <fpage>203 p</fpage>.</citation>
</ref>
<ref id="B107">
<label>107</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bilmes</surname> <given-names>J</given-names>
</name>
</person-group>. <source>Discourse and Behavior</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>1986</year>). <fpage>1 p</fpage>.</citation>
</ref>
<ref id="B108">
<label>108</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kasper</surname> <given-names>G.</given-names>
</name>
</person-group> <article-title>Speech acts in interaction: Towards discursive pragmatics</article-title>. In: <source>Pragmatics &amp; language learning</source>. <person-group person-group-type="editor">
<name>
<surname>Bardori-Harlig</surname> <given-names>K</given-names>
</name>
<name>
<surname>F&#xe9;lix-Brasdefer</surname> <given-names>C</given-names>
</name>
<name>
<surname>Omar</surname> <given-names>A</given-names>
</name>
</person-group>. eds. <publisher-name>National Foreign Language Resource Center</publisher-name>, <publisher-loc>Honolulu, HI</publisher-loc>.</citation>
</ref>
<ref id="B109">
<label>109</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gonz&#xe1;lez-Lloret</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Conversation analysis and speech act performance</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Mart&#xed;nez-Flor</surname> <given-names>A</given-names>
</name>
<name>
<surname>Us&#xf3;-Juan</surname> <given-names>E</given-names>
</name>
</person-group>, editors. <source>Speech Act Performance: Theoretical, empirical and methodological issues</source>. <publisher-name>John Benjamins Publishing Company</publisher-name>, <publisher-loc>Amsterdam</publisher-loc> (<year>2010</year>). p. <fpage>57</fpage>&#x2013;<lpage>74</lpage>. Available at: <uri xlink:href="https://benjamins.com/catalog/lllt.26.04gon">https://benjamins.com/catalog/lllt.26.04gon</uri> (Accessed <access-date>November 27, 2024</access-date>).</citation>
</ref>
<ref id="B110">
<label>110</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rude</surname> <given-names>S</given-names>
</name>
<name>
<surname>Gortner</surname> <given-names>EM</given-names>
</name>
<name>
<surname>Pennebaker</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Language use of depressed and depression-vulnerable college students</article-title>. <source>Cogn Emotion</source>. (<year>2004</year>) <volume>18</volume>:<page-range>1121&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/02699930441000030</pub-id>
</citation>
</ref>
<ref id="B111">
<label>111</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heitz</surname> <given-names>U</given-names>
</name>
<name>
<surname>Cherbuin</surname> <given-names>J</given-names>
</name>
<name>
<surname>Menghini-M&#xfc;ller</surname> <given-names>S</given-names>
</name>
<name>
<surname>Egloff</surname> <given-names>L</given-names>
</name>
<name>
<surname>Ittig</surname> <given-names>S</given-names>
</name>
<name>
<surname>Beck</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Comorbidities in patients with an at-risk mental state and first episode psychosis</article-title>. <source>Eur Psychiatr</source>. (<year>2017</year>) <volume>41</volume>:<page-range>S198&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eurpsy.2017.01.2142</pub-id>
</citation>
</ref>
<ref id="B112">
<label>112</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kosmala</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Exploring the status of filled pauses as pragmatic markers: The role of gaze and gesture</article-title>. <source>P&amp;C</source>. (<year>2022</year>) <volume>29</volume>:<page-range>272&#x2013;96</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1075/pc.21020.kos</pub-id>
</citation>
</ref>
<ref id="B113">
<label>113</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clark</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Using uh and um in spontaneous speaking</article-title>. <source>Cognition</source>. (<year>2002</year>) <volume>84</volume>:<fpage>73</fpage>&#x2013;<lpage>111</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0010-0277(02)00017-3</pub-id>
</citation>
</ref>
<ref id="B114">
<label>114</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Levelt</surname> <given-names>W</given-names>
</name>
</person-group>. <article-title>Monitoring and self-repair in speech</article-title>. <source>Cognition</source>. (<year>1983</year>) <volume>14</volume>:<fpage>41</fpage>&#x2013;<lpage>104</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0010-0277(83)90026-4</pub-id>
</citation>
</ref>
<ref id="B115">
<label>115</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matsumoto</surname> <given-names>K</given-names>
</name>
<name>
<surname>Kircher</surname> <given-names>TTJ</given-names>
</name>
<name>
<surname>Stokes</surname> <given-names>PRA</given-names>
</name>
<name>
<surname>Brammer</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Liddle</surname> <given-names>PF</given-names>
</name>
<name>
<surname>McGuire</surname> <given-names>PK</given-names>
</name>
</person-group>. <article-title>Frequency and neural correlates of pauses in patients with formal thought disorder</article-title>. <source>Front Psychiatry</source>. (<year>2013</year>) <volume>4</volume>:<elocation-id>127/abstract</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2013.00127/abstract</pub-id>
</citation>
</ref>
<ref id="B116">
<label>116</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>DeVault</surname> <given-names>D</given-names>
</name>
<name>
<surname>Georgila</surname> <given-names>K</given-names>
</name>
<name>
<surname>Artstein</surname> <given-names>R</given-names>
</name>
<name>
<surname>Morbini</surname> <given-names>F</given-names>
</name>
<name>
<surname>Traum</surname> <given-names>D</given-names>
</name>
<name>
<surname>Scherer</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>Verbal indicators of psychological distress in interactive dialogue with a virtual human</article-title>, in: <conf-name>Proceedings of the SIGDIAL 2013 Conference</conf-name>, <conf-loc>Metz, France</conf-loc>. pp. <fpage>193</fpage>&#x2013;<lpage>202</lpage>.</citation>
</ref>
<ref id="B117">
<label>117</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Costa</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Silva</surname> <given-names>LFLE</given-names>
</name>
</person-group>. <article-title>Parts of speech and filled pauses in schizophrenia</article-title>. <source>Alfa Rev lingu&#xed;st (S&#xe3;o Jos&#xe9; Rio Preto)</source>. (<year>2023</year>) <volume>67</volume>:<fpage>e16993</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1590/1981-5794-e16993t</pub-id>
</citation>
</ref>
<ref id="B118">
<label>118</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andreasen</surname> <given-names>NC</given-names>
</name>
<name>
<surname>Grove</surname> <given-names>WM</given-names>
</name>
</person-group>. <article-title>Thought, language, and communication in schizophrenia: diagnosis and prognosis</article-title>. <source>Schizophr Bull</source>. (<year>1986</year>) <volume>12</volume>:<page-range>348&#x2013;59</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/schbul/12.3.348</pub-id>
</citation>
</ref>
<ref id="B119">
<label>119</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zamperoni</surname> <given-names>G</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Rossell</surname> <given-names>SL</given-names>
</name>
<name>
<surname>Meyer</surname> <given-names>D</given-names>
</name>
<name>
<surname>Sumner</surname> <given-names>PJ</given-names>
</name>
</person-group>. <article-title>Evidence for the factor structure of formal thought disorder: A systematic review</article-title>. <source>Schizophr Res</source>. (<year>2024</year>) <volume>264</volume>:<page-range>424&#x2013;34</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2024.01.006</pub-id>
</citation>
</ref>
<ref id="B120">
<label>120</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Song</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>H</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Bert-based latent semantic analysis (Bert-LSA): A case study on geospatial data technology and application trend analysis</article-title>. <source>Appl Sci</source>. (<year>2021</year>) <volume>11</volume>:<fpage>11897</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app112411897</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>