<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neuroinform.</journal-id>
<journal-title>Frontiers in Neuroinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neuroinform.</abbrev-journal-title>
<issn pub-type="epub">1662-5196</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fninf.2025.1647194</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Speech pattern disorders in verbally fluent individuals with autism spectrum disorder: a machine learning analysis</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Hu</surname> <given-names>Chuanbo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3100121/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Thrasher</surname> <given-names>Jacob</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Wenqi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ruan</surname> <given-names>Mindi</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yu</surname> <given-names>Xiangxu</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2324916/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Paul</surname> <given-names>Lynn K.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/929695/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Shuo</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1805175/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Li</surname> <given-names>Xin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2730116/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Computer Science, University at Albany</institution>, <addr-line>Albany, NY</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Lane Department of Computer Science and Electrical Engineering, West Virginia University</institution>, <addr-line>Morgantown, WV</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Radiology, Washington University in St. Louis</institution>, <addr-line>St. Louis, MO</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Humanities and Social Sciences, California Institute of Technology</institution>, <addr-line>Pasadena, CA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2694425/overview">Vinoth Jagaroo</ext-link>, Emerson College, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/77580/overview">Stacy Lynn Andersen</ext-link>, Boston University, United States</p><p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3126989/overview">Frederic Briend</ext-link>, INSERM U1253 Imagerie et Cerveau (iBrain), France</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Xin Li <email>xli48&#x00040;albany.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>24</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1647194</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>06</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Hu, Thrasher, Li, Ruan, Yu, Paul, Wang and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Hu, Thrasher, Li, Ruan, Yu, Paul, Wang and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Diagnosing Autism Spectrum Disorder (ASD) in verbally fluent individuals based on speech patterns in examiner-patient dialogues is challenging because speech-related symptoms are often subtle and heterogeneous. This study aimed to identify distinctive speech characteristics associated with ASD by analyzing recorded dialogues from the Autism Diagnostic Observation Schedule (ADOS-2).</p></sec>
<sec>
<title>Methods</title>
<p>We analyzed examiner-participant dialogues from ADOS-2 Module 4 and extracted 40 speech-related features categorized into intonation, volume, rate, pauses, spectral characteristics, chroma, and duration. These acoustic and prosodic features were processed using advanced speech analysis tools and used to train machine learning models to classify ASD participants into two subgroups: those with and without A2-defined speech pattern abnormalities. Model performance was evaluated using cross-validation and standard classification metrics.</p></sec>
<sec>
<title>Results</title>
<p>Using all 40 features, the support vector machine (SVM) achieved an F1-score of 84.49%. After removing Mel-Frequency Cepstral Coefficients (MFCC) and Chroma features to focus on prosodic, rhythmic, energy, and selected spectral features aligned with ADOS-2 A2 scores, performance improved, achieving 85.77% accuracy and an F1-score of 86.27%. Spectral spread and spectral centroid emerged as key features in the reduced set, while MFCC 6 and Chroma 4 also contributed significantly in the full feature set.</p></sec>
<sec>
<title>Discussion</title>
<p>These findings demonstrate that a compact, diverse set of non-MFCC and selected spectral features effectively characterizes speech abnormalities in verbally fluent individuals with ASD. The approach highlights the potential of context-aware, data-driven models to complement clinical assessments and enhance understanding of speech-related manifestations in ASD.</p></sec></abstract>
<kwd-group>
<kwd>speech pattern</kwd>
<kwd>ASD</kwd>
<kwd>machine learning</kwd>
<kwd>ADOS</kwd>
<kwd>audio</kwd>
<kwd>medical dialogues</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="7"/>
<equation-count count="1"/>
<ref-count count="45"/>
<page-count count="13"/>
<word-count count="8571"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value></meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Autism spectrum disorder (ASD) is a developmental condition that presents considerable challenges in social interaction, communication, and behavior (<xref ref-type="bibr" rid="B19">Leekam et al., 2011</xref>; <xref ref-type="bibr" rid="B21">Lord et al., 2018</xref>, <xref ref-type="bibr" rid="B20">2020</xref>). In the United States, ASD affects approximately 1 in 36 children and 1 in 45 adults, making it a critical public health concern (<xref ref-type="bibr" rid="B24">Maenner, 2020</xref>; <xref ref-type="bibr" rid="B9">Dietz et al., 2020</xref>). Despite its prevalence, diagnosing ASD is complex, relying heavily on subjective assessments of behavior and the clinical expertise of specialists. These complexities are compounded by differences in diagnostic standards and healthcare availability across regions, resulting in delayed diagnoses and limiting early intervention opportunities for many families (<xref ref-type="bibr" rid="B8">Daniels and Mandell, 2014</xref>). This subjectivity can lead to inconsistencies in the accuracy and timing of diagnoses across various regions and populations.</p>
<p>ASD diagnosis is traditionally conducted through clinical interviews and behavioral observations, often following standardized tools such as the Autism Diagnostic Observation Schedule (ADOS) (<xref ref-type="bibr" rid="B22">Lord et al., 1999</xref>). ADOS-2 consists of five modules, each tailored to different age groups and language abilities, ranging from nonverbal toddlers to verbally fluent adults. Module 1 is designed for minimally verbal children, Module 2 for those with some phrase speech, Module 3 for verbally fluent children and adolescents, Module 4 for verbally fluent adults, and the Toddler Module for children under 30 months of age. This structured approach allows clinicians to assess social communication, interaction, and restricted or repetitive behaviors across diverse developmental stages. However, these methods require extensive clinician expertise, leading to potential inconsistencies in diagnosis and accessibility issues in underserved areas (<xref ref-type="bibr" rid="B27">Matson and Kozlowski, 2011</xref>; <xref ref-type="bibr" rid="B10">Elsabbagh et al., 2012</xref>). ADOS-2 assessments require trained clinicians who can administer structured tasks, score behavioral responses, and interpret results based on standardized criteria. This specialized training is costly and time-intensive, contributing to a shortage of qualified professionals, especially in regions with limited healthcare resources. Moreover, ASD evaluations are often expensive, requiring multiple clinical visits, making it difficult for families in lower-income communities to access timely assessments. As a result, there is increasing interest in technology-driven approaches that can enhance diagnostic consistency and accessibility (<xref ref-type="bibr" rid="B12">Fletcher-Watson and Happ&#x000E9;, 2019</xref>; <xref ref-type="bibr" rid="B40">Song et al., 2019</xref>; <xref ref-type="bibr" rid="B33">Rezaee, 2025</xref>).</p>
<p>One promising approach is the use of speech analysis for ASD detection. Speech is a fundamental mode of communication, and research suggests that individuals with ASD often exhibit distinctive speech characteristics, including atypical intonation, altered rhythm, abnormal speech rate, and variations in pitch modulation (<xref ref-type="bibr" rid="B29">Mody and Belliveau, 2013</xref>; <xref ref-type="bibr" rid="B31">Pickles et al., 2009</xref>; <xref ref-type="bibr" rid="B41">Vogindroukas et al., 2022</xref>; <xref ref-type="bibr" rid="B26">Martin and Rouas, 2024</xref>). These abnormalities can emerge in development, offering a potential biomarker for ASD diagnosis (<xref ref-type="bibr" rid="B4">Bonneh et al., 2011</xref>). Advances in computational speech processing enable precise analysis of these features, paving the way for non-invasive, scalable, and cost-effective diagnostic tools that could complement existing clinical methods.</p>
<p>Recent advancements in machine learning have further expanded the possibilities for ASD diagnosis by enabling automated detection of behavioral and linguistic patterns (<xref ref-type="bibr" rid="B42">Wang et al., 2015</xref>; <xref ref-type="bibr" rid="B35">Ruan et al., 2021</xref>, <xref ref-type="bibr" rid="B36">2023</xref>; <xref ref-type="bibr" rid="B43">Zhang et al., 2022</xref>). For example, machine learning techniques have been applied to digital behavioral phenotyping (<xref ref-type="bibr" rid="B30">Perochon et al., 2023</xref>) and automated analysis of gestures and facial expressions from video recordings (<xref ref-type="bibr" rid="B18">Lakkapragada et al., 2022</xref>; <xref ref-type="bibr" rid="B17">Krishnappa Babu et al., 2023</xref>). Natural language processing (NLP) has also been applied to electronic health records to derive ASD phenotypes (<xref ref-type="bibr" rid="B45">Zhao et al., 2022</xref>). Speech features are increasingly recognized as digital biomarkers in clinical decision support (<xref ref-type="bibr" rid="B37">Sariyanidi et al., 2025</xref>). Advances in representation learning, such as GANs and self-supervised models, have demonstrated improved ASD speech recognition performance, even in data-limited conditions (<xref ref-type="bibr" rid="B39">Sohn et al., 2025</xref>; <xref ref-type="bibr" rid="B1">Al Futaisi et al., 2025</xref>). On a different scale, <xref ref-type="bibr" rid="B32">Rajagopalan et al. (2024)</xref> showed that robust prediction can be achieved with minimal feature sets across large cohorts, while multi-modal approaches such as facial expression analysis are emerging as valuable complements to speech-based diagnosis (<xref ref-type="bibr" rid="B25">Mahmood et al., 2025</xref>). Building on these successes, leveraging ML for speech analysis offers a promising and relatively unexplored direction in ASD diagnosis.</p>
<p>This study targets verbally fluent individuals assessed with ADOS-2 Module 4 and classifies participants with vs. without A2-defined speech abnormalities. Our goal is not to distinguish ASD from non-ASD; rather, we examine how machine learning can characterize speech-related abnormalities within this subgroup and how such models might complement clinical practice. This research focuses on the following key objectives:</p>
<p><bold>Comprehensive speech feature extraction:</bold> we employed advanced signal processing techniques to extract 40 distinct speech features, grouped into prosodic, rhythmic, spectral, and energy-related categories, to capture subtle ASD-related speech patterns.</p>
<p><bold>Machine learning-based classification:</bold> we applied machine learning models to classify participants with vs. without ADOS-2 A2-defined speech abnormalities, providing an objective framework for analyzing atypical prosody and rhythm.</p>
<p><bold>Complementary clinical insight:</bold> Rather than diagnosing ASD <italic>per se</italic>, this study evaluates whether acoustic speech features can support the characterization of speech abnormalities in verbally fluent individuals with ASD, serving as a data-driven complement to traditional clinical assessments.</p>
<p>This study represents a significant methodological advancement in diagnosis of speech abnormalities in ASD by integrating machine learning with detailed speech analysis. The use of a comprehensive set of speech features, combined with sophisticated machine learning techniques, offers a notable improvement over traditional diagnostic methods. This approach holds the potential for more accurate and earlier detection of ASD, which is critical for timely intervention. Ultimately, the research aims to contribute to personalized treatment and management strategies, enhancing outcomes for individuals with ASD and providing a scalable, objective solution for clinical use. This work focuses on autistic individuals assessed with ADOS-2 Module 4 (verbally fluent adolescents and adults); accordingly, findings pertain to this subgroup rather than the autism spectrum as a whole.</p></sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<sec>
<title>2.1 Caltech audio dataset</title>
<sec>
<title>2.1.1 Autism Diagnostic Observation Schedule (ADOS)</title>
<p>The Autism Diagnostic Observation Schedule, Second Edition (ADOS-2) (<xref ref-type="bibr" rid="B22">Lord et al., 1999</xref>; <xref ref-type="bibr" rid="B2">American Psychiatric Association et al., 2013</xref>) is a widely used standardized instrument for diagnosing ASD. Module 4 of ADOS-2 is specifically designed for verbally fluent adolescents and adults, typically aged 16 and older, and differs from other modules intended for younger or non-verbal individuals. This study focuses on the A2 score, which assesses abnormalities in speech patterns, including intonation, volume, rate, and rhythm. Details for each A2 score level are provided in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Speech abnormalities associated with autism (intonation/volume/rhythm/rate).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Score</bold></th>
<th valign="top" align="left"><bold>Description</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">0</td>
<td valign="top" align="left">Appropriately varying intonation, reasonable volume, and normal rate of speech, with regular rhythm coordinated with breathing.</td>
</tr> <tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Little variation in pitch and tone; rather flat or exaggerated intonation, but not obviously peculiar, OR slightly unusual volume, AND/OR speech that tends to be somewhat unusually slow, fast, or jerky.</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Speech that is clearly abnormal for ANY of the following reasons: slow and halting; inappropriately rapid; jerky and irregular in rhythm (other than ordinary stutter/stammer), such that there is some interference with intelligibility; odd intonation or inappropriate pitch and stress; markedly flat and toneless ("mechanical"); consistently abnormal volume.</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">Stutter or stammer or other fluency disorder (if odd intonation is also present, code 1 or 2 accordingly).</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>2.1.2 ADOS interview audio dataset</title>
<p>The ADOS sessions were conducted sequentially, involving 15 structured scenario tasks designed to elicit responses across a range of communicative and social interactions (see <xref ref-type="table" rid="T2">Table 2</xref>). These tasks allow clinicians to capture meaningful speech and behavioral data, including intonation and speech rate, for analysis. In this study, the Caltech Audio Dataset (<xref ref-type="bibr" rid="B43">Zhang et al., 2022</xref>) includes 33 verbally fluent participants with ASD (26 male, 7 female), aged 16-37 years. The average age of ASD participants was 23.45 &#x000B1; 4.76 years. Nine of these individuals were assessed twice, approximately six months apart, yielding a total of 42 recording sessions. As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, 19 participants exhibited speech abnormalities (A2 &#x02265; 1), while 14 participants received an A2 score of 0. Based on this distribution, the recordings were grouped into ASD with vs. without speech-related abnormalities. To enhance granularity and contextual specificity, each session was further segmented into 15 structured scenario tasks, resulting in 42 &#x000D7; 15 &#x0003D; 630 scenario-level samples, which served as the basic units for subsequent binary classification analyses.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Overview of SCENARIO TASKS in ADOS-2 module 4 diagnosing process.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Scenario</bold></th>
<th valign="top" align="left"><bold>Name</bold></th>
<th valign="top" align="left"><bold>Explanation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>S</italic><sub>1</sub></td>
<td valign="top" align="left">Construction Task</td>
<td valign="top" align="left">Involves the participant engaging in a task that requires constructing or assembling a set structure, testing spatial and motor skills, rather than communicative abilities.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>2</sub></td>
<td valign="top" align="left">Telling a Story from a Book</td>
<td valign="top" align="left">Primarily a monologic task where the participant recounts a story from a book, differing from spontaneous dialogic interactions.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>3</sub></td>
<td valign="top" align="left">Description of a Picture</td>
<td valign="top" align="left">Participants describe a picture, testing their ability to interpret visual information and articulate a coherent description.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>4</sub></td>
<td valign="top" align="left">Conversation and Reporting</td>
<td valign="top" align="left">Focuses on the ability to engage in back-and-forth conversation and to report on past events.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>5</sub></td>
<td valign="top" align="left">Current Work and School</td>
<td valign="top" align="left">Discusses participants&#x00027; current educational and occupational engagements.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>6</sub></td>
<td valign="top" align="left">Social Difficulties and Annoyance</td>
<td valign="top" align="left">Elicits experiences of social challenges and annoyances.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>7</sub></td>
<td valign="top" align="left">Emotions</td>
<td valign="top" align="left">Requires participants to express and identify emotions.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>8</sub></td>
<td valign="top" align="left">Demonstration Task</td>
<td valign="top" align="left">Requires the participant to demonstrate how to use an item or explain a process, which does not involve interactive communication with an examiner.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>9</sub></td>
<td valign="top" align="left">Cartoons</td>
<td valign="top" align="left">Involves interpreting sequences and explaining cartoon strips.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>10</sub></td>
<td valign="top" align="left">Break</td>
<td valign="top" align="left">A pause or intermission in the assessment, involving no communicative or cognitive tasks.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>11</sub></td>
<td valign="top" align="left">Daily Living</td>
<td valign="top" align="left">Covers daily routines and personal care tasks.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>12</sub></td>
<td valign="top" align="left">Friends, Relationships, and Marriage</td>
<td valign="top" align="left">Discusses personal relationships and social norms regarding friendships and marital status.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>13</sub></td>
<td valign="top" align="left">Loneliness</td>
<td valign="top" align="left">Addresses feelings and situations of loneliness and isolation.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>14</sub></td>
<td valign="top" align="left">Plans and Hopes</td>
<td valign="top" align="left">Involves discussing future aspirations and plans.</td>
</tr> <tr>
<td valign="top" align="left"><italic>S</italic><sub>15</sub></td>
<td valign="top" align="left">Creating a Story</td>
<td valign="top" align="left">Tests creative storytelling abilities in an unstructured task.</td>
</tr></tbody>
</table>
</table-wrap>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Distribution of ADOS-2 Module 4 A2 scores across subjects (0 = normal intonation, 1 = mildly atypical intonation, 2 = markedly atypical intonation).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fninf-19-1647194-g0001.tif">
<alt-text>Bar chart illustrating ADOS-2 A2 scores. The horizontal axis shows scores 0, 1, and 2, while the vertical axis indicates the number of subjects. Scores of 0 and 2 have around 14 subjects each, and score 1 has 16 subjects.</alt-text>
</graphic>
</fig>
<p>In addition, although the age range (16-37 years) may overlap with vocal maturation for some participants, we did not explicitly control for or model potential pubertal voice changes. Because our feature set includes acoustic descriptors (e.g., spectral measures), such effects cannot be fully ruled out; we therefore acknowledge this as a limitation and a direction for future, age-stratified analyses.</p></sec></sec>
<sec>
<title>2.2 Feature extraction for identification of autism speech disorder</title>
<p>Feature extraction plays a crucial role in the analysis of speech data, especially in understanding complex disorders like ASD. It involves quantifying various aspects of speech that may reveal traits associated with ASD. For this study, a comprehensive set of speech features was extracted from recorded dialogues, grouped based on their relevance to ASD. Prosodic speech features, including the number of syllables, pauses, rate of speech, articulation rate, speaking duration, original duration, balance, and frequency, were extracted using the &#x0201C;Myprosody&#x0201D; tool (<xref ref-type="bibr" rid="B38">Shahab, 2025</xref>). This tool integrates multiple speech feature extraction methods, providing a detailed analysis of prosodic elements. Additionally, features such as Mel-Frequency Cepstral Coefficients (MFCCs), spectrograms, and chromagrams were extracted using &#x0201C;pyAudioAnalysis&#x0201D; (<xref ref-type="bibr" rid="B15">Giannakopoulos, 2015</xref>), enriching the dataset with diverse audio representations that are essential for analyzing ASD-related speech patterns. These features are described below and summarized in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Detailed categorization of speech features into relevant categories, with explanations and specific feature counts, tailored for comprehensive speech pattern analysis in clinical assessments such as autism.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>No</bold>.</th>
<th valign="top" align="left"><bold>Category</bold></th>
<th valign="top" align="left"><bold>Features</bold></th>
<th valign="top" align="left"><bold>Explanation</bold></th>
<th valign="top" align="center"><bold>&#x00023;</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Intonation</td>
<td valign="top" align="left">Frequency</td>
<td valign="top" align="left">Fundamental frequency, related to the pitch of the voice.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">MFCCs</td>
<td valign="top" align="left">Mel Frequency Cepstral Coefficients, capture timbral aspects that are crucial for intonation.</td>
<td valign="top" align="center">13</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Volume</td>
<td valign="top" align="left">Energy</td>
<td valign="top" align="left">Measures the signal&#x00027;s loudness.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Entropy of Energy</td>
<td valign="top" align="left">Indicates variation in loudness within a frame.</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Rhythm</td>
<td valign="top" align="left">Zero Crossing Rate (ZCR)</td>
<td valign="top" align="left">Reflects the number of times the waveform crosses zero, related to the frequency of the signal.</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Rate</td>
<td valign="top" align="left">Rate of Speech</td>
<td valign="top" align="left">Measures how fast words are spoken.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Number of Syllables</td>
<td valign="top" align="left">Counts the syllables, indicating speech density and pace.</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Pause</td>
<td valign="top" align="left">Number of Pauses</td>
<td valign="top" align="left">Total pauses, reflecting speech interruptions and flow.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Balance</td>
<td valign="top" align="left">Ratio of speaking to pausing, indicates rhythmic flow.</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Spectral</td>
<td valign="top" align="left">Spectral Centroid</td>
<td valign="top" align="left">Center of gravity, affects perceived pitch and sharpness.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Spectral Spread</td>
<td valign="top" align="left">Measures the width of the spectrum, related to the sharpness of sound.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Spectral Rolloff</td>
<td valign="top" align="left">The frequency below which 90% of energy lies, indicates the shape.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Spectral Flux</td>
<td valign="top" align="left">Measures the changes between frames, indicates rhythm changes.</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Spectral Entropy</td>
<td valign="top" align="left">Reflects the entropy of spectral distribution, a complexity measure.</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">Chroma</td>
<td valign="top" align="left">Chroma</td>
<td valign="top" align="left">A set of 12 coefficients each representing a semitone within an octave, used in harmony analysis.</td>
<td valign="top" align="center">12</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">Duration</td>
<td valign="top" align="left">Speaking Duration</td>
<td valign="top" align="left">measure speaking time (excluding fillers and pause)</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">Original Duration</td>
<td valign="top" align="left">measure speaking time (including fillers and pause)</td>
<td valign="top" align="center">1</td>
</tr></tbody>
</table>
</table-wrap>
<p>Each category of features captures different characteristics of speech that are potentially altered in ASD:</p>
<list list-type="simple">
<list-item><p>- <bold>Prosody features</bold> such as pitch (fundamental frequency) variations and speech rate are directly related to the emotional and syntactical aspects of speech, which are often atypical in ASD.</p></list-item>
<list-item><p>- <bold>Energy and Zero Crossing Rate</bold> provide basic information about the speech amplitude and frequency, which are useful for detecting abnormalities in speech loudness and pitch changes.</p></list-item>
<list-item><p>- <bold>Spectral and Chroma features</bold> reflect the quality of sound and harmony in speech. These features are sophisticated and can detect subtleties in speech that are not apparent through simple auditory observation.</p></list-item>
<list-item><p>- <bold>MFCCs and their deltas</bold> offer a robust representation of speech based on the human auditory system&#x00027;s perception of the frequency scales, essential for identifying nuanced discrepancies in how individuals with ASD perceive and produce sounds. By analyzing these features using machine learning models, we aim to identify patterns that are indicative of ASD, thereby assisting in the objective and efficient diagnosis of the disorder.</p></list-item>
</list>
</sec>
<sec>
<title>2.3 Classification models for diagnosis of speech abnormalities in ASD and analysis</title>
<p>To classify ASD-related speech patterns, we employed six machine learning algorithms, selected based on their effectiveness in speech processing and biomedical signal classification. The classification process follows three major stages:</p>
<list list-type="simple">
<list-item><p>(1) Model selection based on suitability for structured and unstructured speech features,</p></list-item>
<list-item><p>(2) Feature selection and optimization to improve performance, and</p></list-item>
<list-item><p>(3) Model interpretability to analyze which speech features contribute most to classification.</p></list-item>
</list>
<sec>
<title>Model selection rationale</title>
<p>Each model was selected based on its unique advantages in handling high-dimensional, speech-derived features:</p>
<list list-type="bullet">
<list-item><p>Support Vector Machine (SVM) (<xref ref-type="bibr" rid="B7">Cortes, 1995</xref>): Works well in high-dimensional spaces and can handle non-linear decision boundaries using Radial Basis Function (RBF) kernels.</p></list-item>
<list-item><p>Random Forest (RF) (<xref ref-type="bibr" rid="B5">Breiman, 2001</xref>): An ensemble learning approach that enhances prediction stability by aggregating multiple decision trees.</p></list-item>
<list-item><p>Gradient Boosting (GB) (<xref ref-type="bibr" rid="B14">Friedman, 2001</xref>): Sequentially builds trees to correct errors of previous iterations, optimizing for complex non-linear relationships.</p></list-item>
<list-item><p>Adaptive Boosting (AdaBoost) (<xref ref-type="bibr" rid="B13">Freund and Schapire, 1997</xref>): Assigns higher weights to misclassified samples, improving generalization while being prone to noise sensitivity.</p></list-item>
<list-item><p>K-Nearest Neighbors (KNN) (<xref ref-type="bibr" rid="B11">Fix and Hodges, 1951</xref>): A distance-based classifier, useful when labels have well-separated clusters in feature space.</p></list-item>
<list-item><p>Na&#x000EF;ve Bayes (NB) (<xref ref-type="bibr" rid="B34">Rish et al., 2001</xref>): A probabilistic model assuming feature independence, known for fast training and robust results in speech applications.</p></list-item>
</list>
<p>Each model was implemented in Python (Scikit-Learn) and trained using 5-fold cross-validation to assess robustness.</p></sec>
<sec>
<title>Hyperparameter tuning</title>
<p>Hyperparameters were optimized using grid search and random search techniques:</p>
<list list-type="bullet">
<list-item><p>Grid Search: Exhaustive search of pre-defined parameter sets for SVM, Random Forest, and Boosting models.</p></list-item>
<list-item><p>Random Search: Used for KNN and AdaBoost, where sampling over parameter space provides efficient exploration.</p></list-item>
</list>
<p>Each model&#x00027;s hyperparameter settings are detailed in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Machine learning models and hyperparameter settings for ASD classification.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left"><bold>Hyperparameters</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">C=0.1, Kernel=RBF, Gamma=scale, Tolerance=1e-3, Max Iterations=-1</td>
</tr> <tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">Trees=100, Max Depth=None, Min Samples Split=10, Min Samples Leaf=5, Bootstrap=True</td>
</tr> <tr>
<td valign="top" align="left">GB</td>
<td valign="top" align="left">Learning Rate=0.1, Trees=100, Max Depth=3, Min Samples Split=5, Subsample=0.8</td>
</tr> <tr>
<td valign="top" align="left">AdaBoost</td>
<td valign="top" align="left">Estimators=50, Learning Rate=1.0, Base Estimator=Decision Stump, Algorithm=SAMME.R</td>
</tr> <tr>
<td valign="top" align="left">KNN</td>
<td valign="top" align="left">K=5, Distance=Euclidean, Weights=Uniform, Algorithm=Auto, Leaf Size=30</td>
</tr> <tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">Distribution=Gaussian, Variance Smoothing=1e-9</td>
</tr></tbody>
</table>
</table-wrap>
<p>The performance of each model was assessed using multiple metrics, including accuracy, precision, recall, and F1-score, calculated through cross-validation across the dataset.</p>
<p>In addition, we employed 5-fold GroupKFold cross-validation to evaluate model performance, ensuring that recordings from the same participant were not split across folds. This choice was made to balance bias and variance in model evaluation, given the limited dataset size.</p></sec></sec>
<sec>
<title>2.4 Feature importance evaluation</title>
<p>To enhance transparency in ASD classification, we applied several interpretability techniques to analyze feature contributions. Shapley Additive Explanations (SHAP) (<xref ref-type="bibr" rid="B23">Lundberg and Lee, 2017</xref>) was employed to estimate the impact of each speech feature on model predictions. SHAP values were computed for all samples, allowing us to examine both individual and global feature influences. SHAP was chosen because it provides consistent, theoretically grounded attributions that are model-agnostic, making it especially suitable for comparing feature relevance across diverse classifiers (e.g., SVM, Random Forest, Gradient Boosting). Alternative methods such as LIME, permutation importance, or partial dependence plots (PDP) were considered; however, SHAP was prioritized due to its ability to capture both local and global interpretability in a unified framework. We acknowledge that SHAP is computationally more expensive than these alternatives, and this aspect is discussed further in the Limitations section. This approach provided insight into how changes in speech characteristics affect classification probability, facilitating a better understanding of model decisions.</p>
<p>For tree-based models such as Random Forest and Gradient Boosting, feature importance was derived using the Mean Decrease in Impurity (MDI) metric. This method ranks features based on their contribution to reducing uncertainty in classification. Additionally, we applied permutation importance to models that do not natively provide feature rankings, such as SVM and KNN. By randomly shuffling each feature and measuring its effect on model performance, we identified the most influential features for ASD classification.</p>
<p>Given that ADOS-2 Module 4 consists of 15 structured tasks, we conducted a scenario-specific feature analysis to investigate whether feature importance varies across different conversational contexts. This analysis involved computing SHAP values separately for each task, allowing us to assess how models rely on specific speech features under varying conditions.</p>
<p>To further interpret model decisions, we incorporated visualization techniques, including SHAP summary plots, feature importance rankings, and scenario-wise importance heatmaps. These visual tools help illustrate patterns in speech-related features and aid in understanding how classification decisions are made. By integrating multiple interpretability methods, we aimed to ensure that our models remain transparent and suitable for potential clinical applications.</p>
<p>The combination of SHAP analysis, feature ranking, and visualization techniques allows for a comprehensive assessment of model behavior. These interpretability methods provide essential insights for refining ASD classification models, validating the consistency of learned patterns, and supporting future improvements in automated diagnostic tools.</p></sec></sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec>
<title>3.1 Experimental setup</title>
<p>To evaluate model performance, we applied a supervised classification framework using the extracted speech features. All experiments were conducted in Python (Scikit-learn) with 5-fold cross-validation to ensure robustness and reduce overfitting. Models were trained and tested on both feature sets described in Section 2.3 (the full 40-feature set and the reduced 15-feature set). We assessed diagnostic performance using four standard classification metrics:</p>
<list list-type="bullet">
<list-item><p><bold>Accuracy</bold>: The proportion of correctly classified samples out of all samples.</p></list-item>
<list-item><p><bold>Precision</bold>: The proportion of predicted positive cases that are true positives, measuring the reliability of positive predictions.</p></list-item>
<list-item><p><bold>Recall (Sensitivity)</bold>: The proportion of true positive cases correctly identified, reflecting the ability to capture actual ASD cases.</p></list-item>
<list-item><p><bold>F1-score</bold>: The harmonic mean of precision and recall, balancing the trade-off between false positives and false negatives.</p></list-item>
</list>
<p>Formally, given true positives (TP), false positives (FP), false negatives (FN), and true negatives (TN):</p>
<disp-formula id="E1"><mml:math id="M1"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>Accuracy</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>Precision</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mtext>&#x000A0;Recall</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;F1&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x02212;</mml:mo><mml:mtext>&#x000A0;score</mml:mtext><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mtext>Precision</mml:mtext><mml:mo>&#x000D7;</mml:mo><mml:mtext>Recall</mml:mtext></mml:mrow><mml:mrow><mml:mtext>Precision</mml:mtext><mml:mo>+</mml:mo><mml:mtext>Recall</mml:mtext></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>These metrics are widely used in medical classification tasks and provide complementary perspectives on diagnostic reliability. Accuracy summarizes overall performance, precision emphasizes avoiding false positives, recall emphasizes capturing true cases, and the F1-score balances both aspects.</p></sec>
<sec>
<title>3.2 Analysis of speech pattern features</title>
<p>To explore the relationships between these features, we calculated Pearson correlation coefficients, measuring the degree and direction of linear relationships (see <xref ref-type="fig" rid="F2">Figure 2</xref>). This approach is crucial for identifying redundancies, interdependencies, and unique contributions of each feature, which can enhance model interpretability and performance by mitigating multicollinearity. Several notable patterns emerge:</p>
<list list-type="bullet">
<list-item><p><bold>High Correlation Among Rate-Based Features:</bold> The rate of speech and articulation rate are strongly correlated, confirming that faster speech naturally leads to a greater number of syllables articulated per unit time. This redundancy suggests that only one of these features may be necessary for robust classification.</p></list-item>
<list-item><p><bold>Duration and Pause-Related Measures:</bold> Speaking duration, original duration, and balance also show moderate-to-strong correlations, reflecting the intertwined nature of fluency, pause frequency, and overall timing. Longer utterances often correspond with proportionally longer pauses, which are captured in the balance measure.</p></list-item>
<list-item><p><bold>Spectral and Prosodic Overlap:</bold> Several spectral features (e.g., spectral spread, centroid, and flux) cluster together, indicating they capture related aspects of energy distribution and spectral sharpness. This suggests potential dimensionality reduction opportunities for spectral descriptors.</p></list-item>
<list-item><p><bold>Zero Crossing Rate (ZCR):</bold> Notably, ZCR exhibits a relatively high correlation with spectral flux and spectral centroid. This indicates that temporal fluctuations in signal polarity are linked to changes in frequency distribution and energy transitions. Since ZCR is a simple yet computationally inexpensive measure, its strong correlation with more complex spectral descriptors suggests it may serve as a lightweight proxy for certain spectral dynamics in ASD-related speech analysis.</p></list-item>
<list-item><p><bold>MFCC and Chroma Clusters:</bold> MFCCs are highly intercorrelated, as expected given their derivation from the same cepstral representation. Similarly, the 12 Chroma features show block-wise correlations, particularly between adjacent chroma bands, reflecting harmonic relationships inherent in speech tonality.</p></list-item>
</list>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Heatmap of Pearson correlation coefficients among all extracted speech features. The color scale represents the strength and direction of correlations (red = strong positive, blue = strong negative).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fninf-19-1647194-g0002.tif">
<alt-text>Heatmap displaying the correlation matrix of various speech features, including Number of Syllables, Speech Rate, and MFCCs, among others. The color scale ranges from -1.0 to 1.0, with red indicating positive correlation and blue indicating negative correlation. A strong diagonal line of red squares shows high self-correlation.</alt-text>
</graphic>
</fig>
<p>These findings highlight redundancy across certain features (e.g., rate measures, MFCCs, Chroma coefficients) as well as unique contributions (e.g., ZCR, spectral spread). This informed our decision to test both a full 40-feature set and a reduced 15-feature set, ensuring that classification models are not unduly biased by collinear predictors.</p></sec>
<sec>
<title>3.3 Classification and analysis of ASD using speech features</title>
<p>In this study, two distinct feature sets were used for classification: (1) all 40 features (including MFCCs and Chroma), and (2) 15 selected features after excluding MFCCs and Chroma. It allows us to assess the necessity of spectral features in ASD detection, especially for cases where computational simplicity is prioritized.</p>
<list list-type="bullet">
<list-item><p>Results with all 40 features: <xref ref-type="table" rid="T5">Table 5</xref> summarizes model performances when using all 40 features. Notably, SVM outperformed other models, achieving the highest F1-score of 84.49%, respectively, underscoring its robustness in capturing nuanced ASD-related speech patterns across a comprehensive feature set.</p></list-item></list>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Comprehensive speech features extracted for analyzing ASD based on 40 features (K = 5).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="center">0.8360 &#x000B1; 0.1334</td>
<td valign="top" align="center"><bold>0.9039</bold> <bold>&#x000B1;</bold> <bold>0.0788</bold></td>
<td valign="top" align="center">0.7974 &#x000B1; 0.1523</td>
<td valign="top" align="center"><bold>0.8449</bold> <bold>&#x000B1;</bold> <bold>0.1192</bold></td>
</tr> <tr>
<td valign="top" align="left">Random Forest</td>
<td valign="top" align="center"><bold>0.8505</bold> <bold>&#x000B1;</bold> <bold>0.0899</bold></td>
<td valign="top" align="center">0.8733 &#x000B1; 0.0481</td>
<td valign="top" align="center">0.8215 &#x000B1; 0.1190</td>
<td valign="top" align="center">0.8423 &#x000B1; 0.0724</td>
</tr> <tr>
<td valign="top" align="left">AdaBoost</td>
<td valign="top" align="center">0.8253 &#x000B1; 0.0815</td>
<td valign="top" align="center">0.8197 &#x000B1; 0.0904</td>
<td valign="top" align="center">0.8190 &#x000B1; 0.1100</td>
<td valign="top" align="center">0.8153 &#x000B1; 0.0842</td>
</tr> <tr>
<td valign="top" align="left">Naive Bayes</td>
<td valign="top" align="center">0.7776 &#x000B1; 0.0906</td>
<td valign="top" align="center">0.7542 &#x000B1; 0.0833</td>
<td valign="top" align="center">0.7800 &#x000B1; 0.1216</td>
<td valign="top" align="center">0.7630 &#x000B1; 0.0885</td>
</tr> <tr>
<td valign="top" align="left">KNN</td>
<td valign="top" align="center">0.8349 &#x000B1; 0.0912</td>
<td valign="top" align="center">0.8153 &#x000B1; 0.0411</td>
<td valign="top" align="center">0.8202 &#x000B1; 0.1232</td>
<td valign="top" align="center">0.8146 &#x000B1; 0.0752</td>
</tr> <tr>
<td valign="top" align="left">Gradient Boosting</td>
<td valign="top" align="center">0.8415 &#x000B1; 0.0751</td>
<td valign="top" align="center">0.8296 &#x000B1; 0.0741</td>
<td valign="top" align="center"><bold>0.8318</bold> <bold>&#x000B1;</bold> <bold>0.1094</bold></td>
<td valign="top" align="center">0.8267 &#x000B1; 0.0750</td>
</tr> <tr>
<td valign="top" align="left">Voting Ensemble</td>
<td valign="top" align="center">0.8503 &#x000B1; 0.0908</td>
<td valign="top" align="center">0.8718 &#x000B1; 0.0538</td>
<td valign="top" align="center">0.8272 &#x000B1; 0.1264</td>
<td valign="top" align="center">0.8442 &#x000B1; 0.0782</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best mean performance within each column; ties are all shown in bold.</p>
</table-wrap-foot>
</table-wrap>
<p>To further justify the choice of 5-fold cross-validation, we directly compared it with 10-fold GroupKFold using the same 40-feature set. As shown in <xref ref-type="table" rid="T5">Tables 5</xref>, <xref ref-type="table" rid="T6">6</xref>, the 5-fold setting yielded slightly higher mean scores in accuracy and F1 score, while also producing consistently smaller standard deviations across nearly all metrics. In contrast, the 10-fold setting led to greater variability, particularly in recall and F1 score, where the standard deviations were substantially larger. This instability is likely due to the smaller test partitions in 10-fold CV, which magnify the impact of sample heterogeneity given our limited dataset size. Taken together, these results indicate that 5-fold CV provides a more stable and reliable estimate of generalization performance in this study, whereas 10-fold CV introduced higher variance and less consistent outcomes.</p>
<list list-type="bullet">
<list-item><p>Results with Selected 15 Features (Excluding MFCCs and Chroma): <xref ref-type="table" rid="T7">Table 7</xref> shows model performances when MFCCs and Chroma features were excluded, resulting in a reduced 15-feature set. The SVM model performed best under this configuration, achieving an accuracy of 85.77% and an F1-score of 86.27%. These results reveal that while spectral features contribute to model accuracy, a simpler feature set without MFCCs and Chroma can still provide competitive performance, making it a viable option for scenarios prioritizing computational efficiency.</p></list-item></list>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Comprehensive speech features extracted for analyzing ASD based on 40 features (K = 10).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="center">0.8313 &#x000B1; 0.1293</td>
<td valign="top" align="center"><bold>0.8993</bold> <bold>&#x000B1;</bold> <bold>0.0782</bold></td>
<td valign="top" align="center">0.7574 &#x000B1; 0.1940</td>
<td valign="top" align="center"><bold>0.8123</bold> <bold>&#x000B1;</bold> <bold>0.1439</bold></td>
</tr> <tr>
<td valign="top" align="left">Random Forest</td>
<td valign="top" align="center"><bold>0.8480</bold> <bold>&#x000B1;</bold> <bold>0.1404</bold></td>
<td valign="top" align="center">0.8194 &#x000B1; 0.2104</td>
<td valign="top" align="center">0.7843 &#x000B1; 0.2151</td>
<td valign="top" align="center">0.7834 &#x000B1; 0.1936</td>
</tr> <tr>
<td valign="top" align="left">AdaBoost</td>
<td valign="top" align="center">0.8415 &#x000B1; 0.1443</td>
<td valign="top" align="center">0.7021 &#x000B1; 0.2568</td>
<td valign="top" align="center">0.7882 &#x000B1; 0.2252</td>
<td valign="top" align="center">0.7309 &#x000B1; 0.2291</td>
</tr> <tr>
<td valign="top" align="left">Naive Bayes</td>
<td valign="top" align="center">0.8058 &#x000B1; 0.1610</td>
<td valign="top" align="center">0.6806 &#x000B1; 0.2494</td>
<td valign="top" align="center">0.7511 &#x000B1; 0.2338</td>
<td valign="top" align="center">0.7064 &#x000B1; 0.2317</td>
</tr> <tr>
<td valign="top" align="left">KNN</td>
<td valign="top" align="center">0.8146 &#x000B1; 0.1440</td>
<td valign="top" align="center">0.6760 &#x000B1; 0.2293</td>
<td valign="top" align="center">0.7692 &#x000B1; 0.2060</td>
<td valign="top" align="center">0.7071 &#x000B1; 0.2007</td>
</tr> <tr>
<td valign="top" align="left">Gradient Boosting</td>
<td valign="top" align="center">0.8446 &#x000B1; 0.1369</td>
<td valign="top" align="center">0.7279 &#x000B1; 0.2541</td>
<td valign="top" align="center"><bold>0.7912</bold> <bold>&#x000B1;</bold> <bold>0.2262</bold></td>
<td valign="top" align="center">0.7396 &#x000B1; 0.2191</td>
</tr> <tr>
<td valign="top" align="left">Voting Ensemble</td>
<td valign="top" align="center">0.8404 &#x000B1; 0.1434</td>
<td valign="top" align="center">0.7698 &#x000B1; 0.2489</td>
<td valign="top" align="center">0.7825 &#x000B1; 0.2191</td>
<td valign="top" align="center">0.7629 &#x000B1; 0.2222</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best mean performance within each column; ties are all shown in bold.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Comprehensive speech features extracted for analyzing ASD without MfCC and chroma.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
<th valign="top" align="center"><bold>F1-Score</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="center"><bold>0.8577</bold> <bold>&#x000B1;</bold> <bold>0.1133</bold></td>
<td valign="top" align="center"><bold>0.9128</bold> <bold>&#x000B1;</bold> <bold>0.0701</bold></td>
<td valign="top" align="center"><bold>0.8213</bold> <bold>&#x000B1;</bold> <bold>0.1351</bold></td>
<td valign="top" align="center"><bold>0.8627</bold> <bold>&#x000B1;</bold> <bold>0.1052</bold></td>
</tr> <tr>
<td valign="top" align="left">Random Forest</td>
<td valign="top" align="center">0.8447 &#x000B1; 0.1078</td>
<td valign="top" align="center">0.8579 &#x000B1; 0.0632</td>
<td valign="top" align="center">0.8195 &#x000B1; 0.1390</td>
<td valign="top" align="center">0.8354 &#x000B1; 0.0987</td>
</tr> <tr>
<td valign="top" align="left">AdaBoost</td>
<td valign="top" align="center">0.8308 &#x000B1; 0.1131</td>
<td valign="top" align="center">0.8217 &#x000B1; 0.0958</td>
<td valign="top" align="center">0.8109 &#x000B1; 0.1476</td>
<td valign="top" align="center">0.8138 &#x000B1; 0.1150</td>
</tr> <tr>
<td valign="top" align="left">KNN</td>
<td valign="top" align="center">0.7993 &#x000B1; 0.1111</td>
<td valign="top" align="center">0.7612 &#x000B1; 0.0882</td>
<td valign="top" align="center">0.7885 &#x000B1; 0.1502</td>
<td valign="top" align="center">0.7715 &#x000B1; 0.1096</td>
</tr> <tr>
<td valign="top" align="left">Gradient Boosting</td>
<td valign="top" align="center">0.7833 &#x000B1; 0.0924</td>
<td valign="top" align="center">0.7553 &#x000B1; 0.0706</td>
<td valign="top" align="center">0.7785 &#x000B1; 0.1288</td>
<td valign="top" align="center">0.7624 &#x000B1; 0.0850</td>
</tr> <tr>
<td valign="top" align="left">Naive Bayes</td>
<td valign="top" align="center">0.7627 &#x000B1; 0.0914</td>
<td valign="top" align="center">0.7084 &#x000B1; 0.0328</td>
<td valign="top" align="center">0.7604 &#x000B1; 0.1374</td>
<td valign="top" align="center">0.7297 &#x000B1; 0.0776</td>
</tr> <tr>
<td valign="top" align="left">Voting Ensemble</td>
<td valign="top" align="center">0.8482 &#x000B1; 0.1179</td>
<td valign="top" align="center">0.8683 &#x000B1; 0.0898</td>
<td valign="top" align="center">0.8209 &#x000B1; 0.1460</td>
<td valign="top" align="center">0.8424 &#x000B1; 0.1183</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best mean performance within each column; ties are all shown in bold.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<title>3.4 Analysis of feature importance</title>
<p>Feature importance analysis was conducted to determine which speech features are most indicative of ASD. The top features were identified based on Mean Decrease in Impurity (MDI) scores from Gradient Boosting for the 40-feature set and permutation importance for SVM in the reduced feature set.</p>
<p>To understand the contributions of each feature in ASD classification, we analyzed feature importance using the SVM model with all 40 features. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the top 10 most important features.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Top 10 important features based on full 40-feature set for ASD classification based on SVM.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fninf-19-1647194-g0003.tif">
<alt-text>Bar chart titled &#x0201C;Top 10 Important Features for SVM&#x0201D; showcasing feature importance. The features, ordered from most to least important, include Spectral Spread, Spectral Centroid, Spectral Flux, Chroma 4, Mfcc 6, Mfcc 8, ZCR, Chroma 9, Energy, and Spectral Entropy. The x-axis represents feature importance values ranging from 0.0000 to 0.0007.</alt-text>
</graphic>
</fig>
<p>In the analysis with the full 40-feature set (as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>), Spectral Spread and Spectral Centroid were the top features, underscoring the importance of spectral distribution in identifying ASD-related speech abnormalities. Spectral Flux and Chroma 4 also contributed significantly, indicating that both spectral energy distribution and pitch variation are relevant for SVM-based classification. The high importance of MFCC 6 for both models highlights its role in capturing timbral aspects of speech that are characteristic of ASD.</p>
<p>With this reduced set (<xref ref-type="fig" rid="F4">Figure 4</xref>), Spectral Spread shows by far the largest average contribution, followed by Spectral Centroid. Spectral Flux also ranks highly, with ZCR contributing to a moderate degree. These results suggest that variation in spectral energy distribution (spread, centroid, flux) constitutes the most informative set of cues for classifying ADOS-2 A2 speech abnormalities in this cohort, with additional contributions from temporal zero-crossing and entropy-based measures.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Top 10 important features based on 15-feature set for ASD classification based on SVM.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fninf-19-1647194-g0004.tif">
<alt-text>Bar chart titled &#x0201C;Top 10 Important Features for SVM&#x0201D;, showing feature importance. Spectral Spread has the highest importance, followed by Spectral Centroid and Spectral Flux. Other features include ZCR, Spectral Entropy, Spectral Rolloff, Frequency, Energy, Articulation Rate, and Speaking Duration. Feature importance is measured on the x-axis.</alt-text>
</graphic>
</fig>
<p>For the scenario-based analysis, we restricted attention to the reduced set of 15 features. This choice was made because (1) the 15-feature set achieved comparable or better performance than the full 40-feature set, and (2) many excluded features (e.g., MFCC, Chroma) are difficult to interpret in clinical or linguistic terms. By focusing on interpretable prosodic and energy-related features, the scenario-level analysis provides insights that are both stable and meaningful for understanding ASD-related speech abnormalities. <xref ref-type="fig" rid="F5">Figure 5</xref> shows the importance of the 15 selected features across the 15 standardized ADOS scenario tasks, highlighting how feature relevance varies with interactional context and enhancing model interpretability.</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Mean feature importance across 15 scenario tasks in ADOS interviews for diagnosis of speech abnormalities in ASD.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fninf-19-1647194-g0005.tif">
<alt-text>Heatmap showing the mean feature importance across different tasks using SHAP values. The vertical axis lists various explanation variables like number of syllables and spectral rolloff. The horizontal axis represents scenario tasks from one to fifteen. Colors range from dark purple indicating low importance to yellow for high importance. A color bar on the right denotes the mean SHAP value scale from zero to point six.</alt-text>
</graphic>
</fig>
<p>From <xref ref-type="fig" rid="F5">Figure 5</xref>, several key patterns emerge. For instance, spectral spread and spectral centroid consistently exhibit relatively high importance across most scenarios, indicating their stability and universal significance in diagnosis of speech abnormalities in ASD across different contexts. Additionally, in Scenario Task 11 and Scenario 13, spectral spread shows particularly high importance, suggesting that these features may capture critical ASD-related speech patterns specific to that task.</p></sec></sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>This study demonstrates that analyzing specific speech patterns in examiner-patient dialogues can significantly aid in diagnosing ASD. By focusing on a comprehensive set of 40 speech-related features, we examined the roles of intonation, volume, rate, pauses, spectral characteristics, Chroma, and duration in distinguishing individuals with ASD. Our findings suggest that a targeted subset of these features&#x02013;primarily prosodic and non-spectral characteristics&#x02013;may offer more effective and computationally efficient diagnostic tools. This discussion addresses the implications of these results, their alignment with existing research, study limitations, and potential directions for future work.</p>
<sec>
<title>4.1 Interpretation of key findings</title>
<p>Our results showed that, while the full feature set achieved strong classification performance, removing MFCC and Chroma features led to an improvement in both accuracy and F1-score. The refined model, focusing on prosodic, rhythmic, and selective spectral features, achieved an accuracy of 85.77% and an F1-score of 86.27%, highlighting the diagnostic potential of simpler, non-spectral features in ASD detection. This improvement underscores the relevance of temporal and prosodic features, such as rate of speech, speaking duration, spectral spread, and frequency, which consistently ranked highly in importance. Notably, spectral spread and frequency emerged as top contributors, supporting the notion that abnormalities in speech fluency, rhythm, and energy distribution are pivotal in ASD-related speech analysis.</p>
<p>The inclusion of the Voting Ensemble further demonstrated that combining multiple classifiers can yield more stable predictions compared to relying on a single model. While SVM achieved the highest mean accuracy and F1-score, its estimates exhibited greater fold-to-fold variability. In contrast, the Voting Ensemble offered a favorable trade-off between accuracy and stability, indicating its potential utility in practical applications where robustness is critical. This highlights the importance of ensemble-based approaches in complementing individual classifiers for speech abnormality detection in ASD.</p>
</sec>
<sec>
<title>4.2 Comparison with previous research</title>
<p>Our approach aligns with existing research that emphasizes the role of prosodic features in identifying ASD-related speech patterns. Previous studies have highlighted irregularities in speech rate, pauses, and intonation as indicative of ASD (<xref ref-type="bibr" rid="B28">McCann and Pepp&#x000E9;, 2003</xref>; <xref ref-type="bibr" rid="B3">Bone et al., 2015</xref>; <xref ref-type="bibr" rid="B16">Holbrook and Israelsen, 2020</xref>). However, our findings extend this by quantitatively demonstrating that reducing reliance on MFCC and Chroma features&#x02013;commonly used in general speech analysis&#x02013;can enhance ASD-specific diagnostic performance. This contrasts with studies that focus heavily on spectral features alone and suggests that a shift toward prosody-based diagnostics may offer a more targeted approach to capturing ASD-related anomalies in speech.</p></sec>
<sec>
<title>4.3 Implications for clinical practice</title>
<p>These findings can assist in the assessment of speech abnormalities in verbally fluent individuals with ASD. While not intended as a stand-alone diagnostic system for ASD, our approach may complement existing clinical practices by providing objective, data-driven measures of prosody and rhythm abnormalities.</p>
<p>Although spectral features alone have achieved high accuracy in prior studies (<xref ref-type="bibr" rid="B6">Briend et al., 2023</xref>), our results highlight that non-spectral features also capture clinically interpretable aspects of prosody and rhythm that are directly relevant to the ADOS-2 A2 assessment. Rather than replacing spectral features, non-spectral features offer complementary value by improving interpretability and aligning closely with clinical constructs.</p>
<p>Furthermore, the identified importance of features tied to ADOS-2 Module 4, specifically the A2 score, underscores the potential for automated analyses to complement clinical assessment by providing objective, data-driven measures of speech abnormalities. This aligns with recent calls for more objective, data-driven approaches in diagnosis of speech abnormalities in ASD to mitigate subjectivity in clinical practice (<xref ref-type="bibr" rid="B44">Zhang and Li, 2024</xref>).</p>
<p>Our scenario-based feature importance analysis (<xref ref-type="fig" rid="F5">Figure 5</xref>, <xref ref-type="table" rid="T2">Table 2</xref>) demonstrates that the diagnostic contribution of speech features is not uniform across tasks. Spectral-domain measures, particularly <italic>Spectral Spread</italic>, consistently emerge as more influential than prosodic timing variables, but their relevance fluctuates depending on the interactional context. For instance, heightened importance of spectral features in scenarios such as S11 (<italic>Daily Living</italic>) and S13 (<italic>Loneliness</italic>) suggests that tasks prompting extended, personally framed, or socially complex responses may accentuate acoustic variability. These context-sensitive effects highlight the value of considering task demands when interpreting speech abnormalities in ASD, and they point toward the development of context-aware diagnostic models.</p>
</sec>
<sec>
<title>4.4 Limitations and future work</title>
<p>Despite promising results, this study has several limitations. First, the dataset&#x00027;s size and demographic characteristics may limit generalizability, as it was based on specific examiner-patient interactions within the ADOS-2 framework. Further studies with larger, more diverse samples are necessary to validate the findings across different populations and settings. Additionally, while this study focused on specific speech features, there may be other relevant variables, such as linguistic content and contextual information, which could enhance diagnostic accuracy if integrated with the current model.</p>
<p>Building on this study, future research could explore integrating additional multimodal data sources, such as facial expressions, gestures, and gaze, which may complement speech patterns in ASD diagnosis. Such a multimodal approach could provide a more holistic view of communicative behaviors associated with ASD, potentially enhancing the accuracy and robustness of diagnostic models.</p>
<p>Another limitation is the gender imbalance in our dataset (26 male vs. 7 female participants). This reflects the higher reported prevalence of ASD in males compared to females, which is consistent with prior epidemiological findings. However, the small number of female participants limits the ability to draw strong conclusions about whether the observed speech-related patterns generalize across genders. It is possible that prosodic and spectral features related to ASD manifest differently in female participants, an aspect that our current dataset is underpowered to investigate. Future research with more balanced cohorts will be essential to examine potential gender-specific differences in ASD-related speech characteristics and to improve the generalizability of diagnostic models.</p>
<p>In addition, another limitation relates to repeated ADOS-2 sessions in a subset of participants. Approximately 20% of the recordings came from follow-up sessions conducted about six months apart with the same individuals. While these sessions captured different conversational content and thus provided valuable within-subject variability, they also introduced potential non-independence of samples. We did not explicitly model or control for this in the present analysis, which may have influenced the stability of the classification results. Future research should address this by using larger independent cohorts or by applying statistical approaches such as mixed-effects modeling to account for repeated measures.</p>
<p>Another limitation concerns our feature reduction strategy. We focused on a theoretically motivated subset of 15 features, excluding MFCC and Chroma coefficients because of their limited interpretability in the context of ASD-related speech abnormalities. While this choice resulted in slightly improved model performance, it was not a fully data-driven reduction. Future studies could incorporate systematic feature selection methods (e.g., recursive feature elimination, LASSO regularization, or correlation-based filtering) to more rigorously identify and remove uninformative features from the full set of 40 features, potentially leading to further performance gains.</p>
<p>Another limitation concerns the relatively large standard deviations observed in some models (e.g., SVM), which reflect variability across cross-validation folds. This variability likely stems from the modest dataset size and the heterogeneity of speech samples across participants. As a result, model performance may be sensitive to how training and test sets are partitioned. Future research with larger and more balanced datasets will be crucial for improving the stability and generalizability of the models.</p>
<p>Moreover, as the study found variations in feature importance across different scenario tasks, developing context-sensitive models could yield further improvements. By tailoring feature weighting or selection to specific social interaction scenarios, future models could better capture the nuanced ways in which ASD manifests across diverse contexts. Additionally, exploring reinforcement learning or other adaptive learning techniques could help create models that dynamically adjust to individual differences in ASD presentations.</p>
<p>Another limitation is that our dataset included only individuals assessed with ADOS-2 Module 4, which is restricted to verbally fluent participants. Consequently, the results may not generalize to minimally verbal or non-verbal autistic individuals. Future research should extend this approach to other ADOS modules to capture a broader range of the autism spectrum. Additionally, the participant age range (16-37 years) spans adolescence and early adulthood, which may include individuals undergoing vocal maturation. We did not explicitly control for or analyze the potential impact of pubertal voice changes on extracted speech features. As a result, vocal maturation could have introduced additional variability in the data, which should be examined in future research with larger and more age-stratified samples.</p>
<p>In conclusion, this study underscores the potential of prosody-based and scenario-sensitive approaches in diagnosis of speech abnormalities in ASD. By reducing reliance on spectral features and leveraging context-specific analysis, future diagnostic tools may become more precise and accessible, supporting earlier and more objective ASD assessments.</p></sec></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>This study demonstrates that analyzing a targeted set of speech features, particularly prosodic and non-spectral characteristics, can effectively support diagnosis of speech abnormalities in ASD. By examining 40 distinct speech features from examiner-patient dialogues, we identified a reduced feature set focused on prosodic and rhythmic attributes, achieving strong diagnostic accuracy and outperforming models that rely on more complex spectral features. The identified features, such as spectral spread, Spectral Centroid, and Spectral Flux, underscore the relevance of non-spectral cues in capturing ASD-related communication patterns.</p>
<p>These findings suggest that a prosody-focused, streamlined approach can enhance accessibility and efficiency in ASD diagnostics. The performance of the reduced feature set highlights its potential for real-time assessments, supporting quicker and more objective screening for speech abnormalities in ASD. Moving forward, integrating context-sensitive models and multimodal data sources could refine and advance ASD diagnostics, ultimately contributing to improved intervention strategies for individuals on the autism spectrum.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/cbhu523/speech_ASD">https://github.com/cbhu523/speech_ASD</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>All participants in this study gave informed consent for participation and data-sharing under a protocol approved by the institutional review board of the California Institute of Technology. Studies conducted with this human data were conducted in accordance with protocols approved by the institutional review boards at California Institute of Technology, WUSTL and UAlbany. The studies were conducted in accordance with the local legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>CH: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. JT: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Writing &#x02013; review &#x00026; editing. WL: Data curation, Writing &#x02013; review &#x00026; editing. MR: Data curation, Writing &#x02013; review &#x00026; editing. XY: Data curation, Writing &#x02013; review &#x00026; editing. LP: Data curation, Validation, Writing &#x02013; review &#x00026; editing. SW: Funding acquisition, Project administration, Supervision, Validation, Writing &#x02013; review &#x00026; editing. XL: Funding acquisition, Project administration, Resources, Supervision, Validation, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This project was partially supported by the National Science Foundation under Grant No. HCC-2401748 and Prof. Xin Li&#x00027;s start-up funds from UAlbany.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al Futaisi</surname> <given-names>N.</given-names></name> <name><surname>Schuller</surname> <given-names>B. W.</given-names></name> <name><surname>Ringeval</surname> <given-names>F.</given-names></name> <name><surname>Pantic</surname> <given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>The noor project: fair transformer transfer learning for autism spectrum disorder recognition from speech</article-title>. <source>Front. Digital Health</source> <volume>7</volume>:<fpage>1274675</fpage>. <pub-id pub-id-type="doi">10.3389/fdgth.2025.1274675</pub-id><pub-id pub-id-type="pmid">40901408</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><collab>American Psychiatric Association</collab></person-group>. (<year>2013</year>). <source>Diagnostic and statistical manual of mental disorders: DSM-5, volume 5</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Psychiatric Association</publisher-name>.</citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bone</surname> <given-names>D.</given-names></name> <name><surname>Black</surname> <given-names>M. P.</given-names></name> <name><surname>Ramakrishna</surname> <given-names>A.</given-names></name> <name><surname>Grossman</surname> <given-names>R. B.</given-names></name> <name><surname>Narayanan</surname> <given-names>S. S.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Acoustic-prosodic correlates of &#x00027;awkward&#x00027;prosody in story retellings from adolescents with autism,&#x0201D;</article-title> in <source>Interspeech</source> (<publisher-loc>Dresden</publisher-loc>: <publisher-name>International Speech Communication Association (ISCA</publisher-name>)), <fpage>1616</fpage>&#x02013;<lpage>1620</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bonneh</surname> <given-names>Y. S.</given-names></name> <name><surname>Levanon</surname> <given-names>Y.</given-names></name> <name><surname>Dean-Pardo</surname> <given-names>O.</given-names></name> <name><surname>Lossos</surname> <given-names>L.</given-names></name> <name><surname>Adini</surname> <given-names>Y.</given-names></name></person-group> (<year>2011</year>). <article-title>Abnormal speech spectrum and increased pitch variability in young autistic children</article-title>. <source>Front. Hum. Neurosci</source>. <volume>4</volume>:<fpage>237</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2010.00237</pub-id><pub-id pub-id-type="pmid">21267429</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name></person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn</source>. <volume>45</volume>, <fpage>5</fpage>&#x02013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Briend</surname> <given-names>F.</given-names></name> <name><surname>David</surname> <given-names>C.</given-names></name> <name><surname>Silleresi</surname> <given-names>S.</given-names></name> <name><surname>Malvy</surname> <given-names>J.</given-names></name> <name><surname>Ferr&#x000E9;</surname> <given-names>S.</given-names></name> <name><surname>Latinus</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Voice acoustics allow classifying autism spectrum disorder with high accuracy</article-title>. <source>Transl. Psychiatry</source> <volume>13</volume>:<fpage>250</fpage>. <pub-id pub-id-type="doi">10.1038/s41398-023-02554-8</pub-id><pub-id pub-id-type="pmid">37422467</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cortes</surname> <given-names>C.</given-names></name></person-group> (<year>1995</year>). <article-title>Support-vector networks</article-title>. <source>Mach. Learn</source>. <volume>20</volume>, <fpage>273</fpage>&#x02013;<lpage>297</lpage>. <pub-id pub-id-type="doi">10.1023/A:1022627411411</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Daniels</surname> <given-names>A. M.</given-names></name> <name><surname>Mandell</surname> <given-names>D. S.</given-names></name></person-group> (<year>2014</year>). <article-title>Explaining differences in age at autism spectrum disorder diagnosis: a critical review</article-title>. <source>Autism</source> <volume>18</volume>, <fpage>583</fpage>&#x02013;<lpage>597</lpage>. <pub-id pub-id-type="doi">10.1177/1362361313480277</pub-id><pub-id pub-id-type="pmid">23787411</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dietz</surname> <given-names>P. M.</given-names></name> <name><surname>Rose</surname> <given-names>C. E.</given-names></name> <name><surname>McArthur</surname> <given-names>D.</given-names></name> <name><surname>Maenner</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>National and state estimates of adults with autism spectrum disorder</article-title>. <source>J. Autism Dev. Disord</source>. <volume>50</volume>, <fpage>4258</fpage>&#x02013;<lpage>4266</lpage>. <pub-id pub-id-type="doi">10.1007/s10803-020-04494-4</pub-id><pub-id pub-id-type="pmid">32390121</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elsabbagh</surname> <given-names>M.</given-names></name> <name><surname>Divan</surname> <given-names>G.</given-names></name> <name><surname>Koh</surname> <given-names>Y.-J.</given-names></name> <name><surname>Kim</surname> <given-names>Y. S.</given-names></name> <name><surname>Kauchali</surname> <given-names>S.</given-names></name> <name><surname>Marc&#x000ED;n</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Global prevalence of autism and other pervasive developmental disorders</article-title>. <source>Autism Res</source>. <volume>5</volume>:<fpage>160</fpage>&#x02013;<lpage>179</lpage>. <pub-id pub-id-type="doi">10.1002/aur.239</pub-id><pub-id pub-id-type="pmid">22495912</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fix</surname> <given-names>E.</given-names></name> <name><surname>Hodges</surname> <given-names>J. L.</given-names></name></person-group> (<year>1951</year>). <article-title>&#x0201C;Discriminatory analysis: Nonparametric discrimination: Consistency properties,&#x0201D;</article-title> in <source>Technical Report No. 4, USAF School of Aviation Medicine</source> (<publisher-loc>Randolph Field, TX</publisher-loc>: <publisher-name>USAF School of Aviation Medicine</publisher-name>).</citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fletcher-Watson</surname> <given-names>S.</given-names></name> <name><surname>Happ&#x000E9;</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <source>Autism: A New Introduction to Psychological Theory and Current Debate</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Routledge</publisher-name>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Freund</surname> <given-names>Y.</given-names></name> <name><surname>Schapire</surname> <given-names>R. E.</given-names></name></person-group> (<year>1997</year>). <article-title>A decision-theoretic generalization of on-line learning and an application to boosting</article-title>. <source>J. Comp. Syst. Sci</source>. <volume>55</volume>, <fpage>119</fpage>&#x02013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1006/jcss.1997.1504</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friedman</surname> <given-names>J. H.</given-names></name></person-group> (<year>2001</year>). <article-title>Greedy function approximation: a gradient boosting machine</article-title>. <source>Ann. Statist</source>. <volume>29</volume>, <fpage>1189</fpage>&#x02013;<lpage>1232</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1013203451</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Giannakopoulos</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). <article-title>Pyaudioanalysis: An open-source python library for audio signal analysis</article-title>. <source>PLoS ONE</source> <volume>10</volume>:<fpage>e0144610</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0144610</pub-id><pub-id pub-id-type="pmid">26656189</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Holbrook</surname> <given-names>S.</given-names></name> <name><surname>Israelsen</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Speech prosody interventions for persons with autism spectrum disorders: a systematic review</article-title>. <source>Am. J. Speech-Lang. Pathol</source>. <volume>29</volume>, <fpage>2189</fpage>&#x02013;<lpage>2205</lpage>. <pub-id pub-id-type="doi">10.1044/2020_AJSLP-19-00127</pub-id><pub-id pub-id-type="pmid">32757615</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krishnappa Babu</surname> <given-names>P. R.</given-names></name> <name><surname>Di Martino</surname> <given-names>J. M.</given-names></name> <name><surname>Chang</surname> <given-names>Z.</given-names></name> <name><surname>Perochon</surname> <given-names>S.</given-names></name> <name><surname>Aiello</surname> <given-names>R.</given-names></name> <name><surname>Carpenter</surname> <given-names>K. L.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Complexity analysis of head movements in autistic toddlers</article-title>. <source>J. Child Psychol. Psychiat</source>. <volume>64</volume>, <fpage>156</fpage>&#x02013;<lpage>166</lpage>. <pub-id pub-id-type="doi">10.1111/jcpp.13681</pub-id><pub-id pub-id-type="pmid">35965431</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lakkapragada</surname> <given-names>A.</given-names></name> <name><surname>Kline</surname> <given-names>A.</given-names></name> <name><surname>Mutlu</surname> <given-names>O. C.</given-names></name> <name><surname>Paskov</surname> <given-names>K.</given-names></name> <name><surname>Chrisman</surname> <given-names>B.</given-names></name> <name><surname>Stockham</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>The classification of abnormal hand movement to aid in autism detection: machine learning study</article-title>. <source>JMIR Biomed. Eng</source>. <volume>7</volume>:<fpage>e33771</fpage>. <pub-id pub-id-type="doi">10.2196/33771</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leekam</surname> <given-names>S. R.</given-names></name> <name><surname>Prior</surname> <given-names>M. R.</given-names></name> <name><surname>Uljarevic</surname> <given-names>M.</given-names></name></person-group> (<year>2011</year>). <article-title>Restricted and repetitive behaviors in autism spectrum disorders: a review of research in the last decade</article-title>. <source>Psychol. Bull</source>. 137, 562. <pub-id pub-id-type="doi">10.1037/a0023341</pub-id><pub-id pub-id-type="pmid">21574682</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lord</surname> <given-names>C.</given-names></name> <name><surname>Brugha</surname> <given-names>T. S.</given-names></name> <name><surname>Charman</surname> <given-names>T.</given-names></name> <name><surname>Cusack</surname> <given-names>J.</given-names></name> <name><surname>Dumas</surname> <given-names>G.</given-names></name> <name><surname>Frazier</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Autism spectrum disorder</article-title>. <source>Nat. Rev. Dis. Prim</source>. <volume>6</volume>, <fpage>1</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1038/s41572-019-0138-4</pub-id><pub-id pub-id-type="pmid">31949163</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lord</surname> <given-names>C.</given-names></name> <name><surname>Elsabbagh</surname> <given-names>M.</given-names></name> <name><surname>Baird</surname> <given-names>G.</given-names></name> <name><surname>Veenstra-Vanderweele</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>Autism spectrum disorder</article-title>. <source>Lancet</source> <volume>392</volume>:<fpage>508</fpage>&#x02013;<lpage>520</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(18)31129-2</pub-id><pub-id pub-id-type="pmid">30078460</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lord</surname> <given-names>C.</given-names></name> <name><surname>Rutter</surname> <given-names>M.</given-names></name> <name><surname>DiLavore</surname> <given-names>P. C.</given-names></name> <name><surname>Risi</surname> <given-names>S.</given-names></name> <name><surname>Gotham</surname> <given-names>K.</given-names></name> <name><surname>Bishop</surname> <given-names>S. L.</given-names></name> <etal/></person-group>. (<year>1999</year>). <source>Ados. Autism Diagnostic Observation Schedule</source>. <publisher-loc>Manual. Los Angeles</publisher-loc>: <publisher-name>WPS</publisher-name>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lundberg</surname> <given-names>S. M.</given-names></name> <name><surname>Lee</surname> <given-names>S.-I.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;A unified approach to interpreting model predictions,&#x0201D;</article-title> in <source>Proceedings of the 31st International Conference on Neural Information Processing Systems (NeurIPS)</source> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates Inc</publisher-name>), <fpage>4768</fpage>&#x02013;<lpage>4777</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maenner</surname> <given-names>M. J.</given-names></name></person-group> (<year>2020</year>). <article-title>Prevalence of autism spectrum disorder among children aged 8 years&#x02014;autism and developmental disabilities monitoring network, 11 sites, United States, 2016. MMWR</article-title>. <source>Surveil. Summar</source>. <volume>69</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.15585/mmwr.ss6904a1</pub-id><pub-id pub-id-type="pmid">32214087</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mahmood</surname> <given-names>M. A.</given-names></name> <name><surname>Jamel</surname> <given-names>L.</given-names></name> <name><surname>Alturki</surname> <given-names>N.</given-names></name> <name><surname>Tawfeek</surname> <given-names>M. A.</given-names></name></person-group> (<year>2025</year>). <article-title>Leveraging artificial intelligence for diagnosis of children autism through facial expressions</article-title>. <source>Sci. Rep</source>. <volume>15</volume>:<fpage>11945</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-025-96014-6</pub-id><pub-id pub-id-type="pmid">40200029</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Martin</surname> <given-names>V. P.</given-names></name> <name><surname>Rouas</surname> <given-names>J.-L.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Why voice biomarkers of psychiatric disorders are not used in clinical practice? deconstructing the myth of the need for objective diagnosis,&#x0201D;</article-title> in <source>Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)</source> (<publisher-loc>Torino</publisher-loc>: <publisher-name>European Language Resources Association (ELRA</publisher-name>)), <fpage>17603</fpage>&#x02013;<lpage>17613</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Matson</surname> <given-names>J. L.</given-names></name> <name><surname>Kozlowski</surname> <given-names>A. M.</given-names></name></person-group> (<year>2011</year>). <article-title>The increasing prevalence of autism spectrum disorders</article-title>. <source>Res. Autism Spectr. Disord</source>. <volume>5</volume>, <fpage>418</fpage>&#x02013;<lpage>425</lpage>. <pub-id pub-id-type="doi">10.1016/j.rasd.2010.06.004</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McCann</surname> <given-names>J.</given-names></name> <name><surname>Pepp&#x000E9;</surname> <given-names>S.</given-names></name></person-group> (<year>2003</year>). <article-title>Prosody in autism spectrum disorders: a critical review</article-title>. <source>Int. J. Lang. Commun. Disord</source>. <volume>38</volume>, <fpage>325</fpage>&#x02013;<lpage>350</lpage>. <pub-id pub-id-type="doi">10.1080/1368282031000154204</pub-id><pub-id pub-id-type="pmid">14578051</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mody</surname> <given-names>M.</given-names></name> <name><surname>Belliveau</surname> <given-names>J. W.</given-names></name></person-group> (<year>2013</year>). <article-title>Speech and language impairments in autism: insights from behavior and neuroimaging</article-title>. <source>North Am. J. Med. Sci</source>. <volume>5</volume>:<fpage>157</fpage>. <pub-id pub-id-type="doi">10.7156/v5i3p157</pub-id><pub-id pub-id-type="pmid">24349628</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Perochon</surname> <given-names>S.</given-names></name> <name><surname>Di Martino</surname> <given-names>J. M.</given-names></name> <name><surname>Carpenter</surname> <given-names>K. L.</given-names></name> <name><surname>Compton</surname> <given-names>S.</given-names></name> <name><surname>Davis</surname> <given-names>N.</given-names></name> <name><surname>Eichner</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Early detection of autism using digital behavioral phenotyping</article-title>. <source>Nat. Med</source>. <volume>29</volume>, <fpage>2489</fpage>&#x02013;<lpage>2497</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-023-02574-3</pub-id><pub-id pub-id-type="pmid">37783967</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pickles</surname> <given-names>A.</given-names></name> <name><surname>Simonoff</surname> <given-names>E.</given-names></name> <name><surname>Conti-Ramsden</surname> <given-names>G.</given-names></name> <name><surname>Falcaro</surname> <given-names>M.</given-names></name> <name><surname>Simkin</surname> <given-names>Z.</given-names></name> <name><surname>Charman</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Loss of language in early development of autism and specific language impairment</article-title>. <source>J. Child Psychol. Psychiat</source>. <volume>50</volume>, <fpage>843</fpage>&#x02013;<lpage>852</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-7610.2008.02032.x</pub-id><pub-id pub-id-type="pmid">19527315</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rajagopalan</surname> <given-names>S. S.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Yahia</surname> <given-names>A.</given-names></name> <name><surname>Tammimies</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Machine learning prediction of autism spectrum disorder from a minimal set of medical and background information</article-title>. <source>JAMA Network Open</source> <volume>7</volume>:<fpage>e2429229</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.29229</pub-id><pub-id pub-id-type="pmid">39158907</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rezaee</surname> <given-names>K.</given-names></name></person-group> (<year>2025</year>). <article-title>Machine learning in automated diagnosis of autism spectrum disorder: a comprehensive review</article-title>. <source>Comp. Sci. Rev</source>. <volume>56</volume>:<fpage>100730</fpage>. <pub-id pub-id-type="doi">10.1016/j.cosrev.2025.100730</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rish</surname> <given-names>I.</given-names></name></person-group> (<year>2001</year>). <article-title>&#x0201C;An empirical study of the naive bayes classifier,&#x0201D;</article-title> in <source>IJCAI 2001 Workshop on Empirical Methods in Artificial Intelligence</source> (<publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>IJCAI</publisher-name>), <fpage>41</fpage>&#x02013;<lpage>46</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ruan</surname> <given-names>M.</given-names></name> <name><surname>Webster</surname> <given-names>P. J.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep neural network reveals the world of autism from a first-person perspective</article-title>. <source>Autism Res</source>. <volume>14</volume>, <fpage>333</fpage>&#x02013;<lpage>342</lpage>. <pub-id pub-id-type="doi">10.1002/aur.2376</pub-id><pub-id pub-id-type="pmid">32869953</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ruan</surname> <given-names>M.</given-names></name> <name><surname>Yu</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>N.</given-names></name> <name><surname>Hu</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Video-based contrastive learning on decision trees: From action recognition to autism diagnosis,&#x0201D;</article-title> in <source>Proceedings of the 14th Conference on ACM Multimedia Systems</source> (<publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>Wiley</publisher-name>), <fpage>289</fpage>&#x02013;<lpage>300</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sariyanidi</surname> <given-names>E.</given-names></name> <name><surname>Yankowitz</surname> <given-names>L.</given-names></name> <name><surname>Schultz</surname> <given-names>R. T.</given-names></name> <name><surname>Herrington</surname> <given-names>J. D.</given-names></name> <name><surname>Tunc</surname> <given-names>B.</given-names></name> <name><surname>Cohn</surname> <given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>Beyond facs: Data-driven facial expression dictionaries, with application to predicting autism</article-title>. <source>arXiv</source> [preprint] arXiv:2505.24679. <pub-id pub-id-type="doi">10.1109/FG61629.2025.11099288</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Shahab</surname> <given-names>K.</given-names></name></person-group> (<year>2025</year>). <source>Myprosody: A Python Library for Measuring Acoustic Features of Speech</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/Shahabks/myprosody">https://github.com/Shahabks/myprosody</ext-link> (Accessed October 13, 2025).</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sohn</surname> <given-names>J.-S.</given-names></name> <name><surname>Lee</surname> <given-names>E.</given-names></name> <name><surname>Kim</surname> <given-names>J.-J.</given-names></name> <name><surname>Oh</surname> <given-names>H.-K.</given-names></name> <name><surname>Kim</surname> <given-names>E.</given-names></name></person-group> (<year>2025</year>). <article-title>Implementation of generative ai for the assessment and treatment of autism spectrum disorders: a scoping review</article-title>. <source>Front. Psychiatry</source> <volume>16</volume>:<fpage>1628216</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyt.2025.1628216</pub-id><pub-id pub-id-type="pmid">40766925</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>D.-Y.</given-names></name> <name><surname>Kim</surname> <given-names>S. Y.</given-names></name> <name><surname>Bong</surname> <given-names>G.</given-names></name> <name><surname>Kim</surname> <given-names>J. M.</given-names></name> <name><surname>Yoo</surname> <given-names>H. J.</given-names></name></person-group> (<year>2019</year>). <article-title>The use of artificial intelligence in screening and diagnosis of autism spectrum disorder: a literature review</article-title>. <source>J. Korean Acad. Child Adolesc. Psychiat</source>. <volume>30</volume>:<fpage>145</fpage>. <pub-id pub-id-type="doi">10.5765/jkacap.190027</pub-id><pub-id pub-id-type="pmid">32595335</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vogindroukas</surname> <given-names>I.</given-names></name> <name><surname>Stankova</surname> <given-names>M.</given-names></name> <name><surname>Chelas</surname> <given-names>E.-N.</given-names></name> <name><surname>Proedrou</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>Language and speech characteristics in autism</article-title>. <source>Neuropsychiatric Dis. Treatm</source>. <volume>2022</volume>, <fpage>2367</fpage>&#x02013;<lpage>2377</lpage>. <pub-id pub-id-type="doi">10.2147/NDT.S331987</pub-id><pub-id pub-id-type="pmid">36268264</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Jiang</surname> <given-names>M.</given-names></name> <name><surname>Duchesne</surname> <given-names>X. M.</given-names></name> <name><surname>Laugeson</surname> <given-names>E. A.</given-names></name> <name><surname>Kennedy</surname> <given-names>D. P.</given-names></name> <name><surname>Adolphs</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Atypical visual saliency in autism spectrum disorder quantified through model-based eye tracking</article-title>. <source>Neuron</source> <volume>88</volume>, <fpage>604</fpage>&#x02013;<lpage>616</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2015.09.042</pub-id><pub-id pub-id-type="pmid">26593094</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>N.</given-names></name> <name><surname>Ruan</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Paul</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>Discriminative few shot learning of facial dynamics in interview videos for autism trait classification</article-title>. <source>IEEE Trans. Affect. Comp</source>. <volume>14</volume>, <fpage>1110</fpage>&#x02013;<lpage>1124</lpage>. <pub-id pub-id-type="doi">10.1109/TAFFC.2022.3178946</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Objective approaches to autism diagnosis: Challenges and opportunities,&#x0201D;</article-title> in <source>Proceedings of the 2024 International Conference on Language Resources and Evaluation (LREC)</source> (<publisher-loc>Luxembourg</publisher-loc>: <publisher-name>European Language Resources Association</publisher-name>), <fpage>1531</fpage>&#x02013;<lpage>1540</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>M.</given-names></name> <name><surname>Havrilla</surname> <given-names>J.</given-names></name> <name><surname>Peng</surname> <given-names>J.</given-names></name> <name><surname>Drye</surname> <given-names>M.</given-names></name> <name><surname>Fecher</surname> <given-names>M.</given-names></name> <name><surname>Guthrie</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Development of a phenotype ontology for autism spectrum disorder by natural language processing on electronic health records</article-title>. <source>J. Neurodev. Disord</source>. <volume>14</volume>:<fpage>32</fpage>. <pub-id pub-id-type="doi">10.1186/s11689-022-09442-0</pub-id><pub-id pub-id-type="pmid">35606697</pub-id></citation></ref>
</ref-list>
</back>
</article>