<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1660388</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Demographic identification of Greater Caribbean manatees via acoustic feature learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Merchan</surname> <given-names>Fernando</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/2770443"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Contreras</surname> <given-names>Kenji</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<uri xlink:href="https://loop.frontiersin.org/people/2807387"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Poveda</surname> <given-names>H&#x000E9;ctor</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1062742"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Est&#x000E9;vez</surname> <given-names>Roc&#x000ED;o M.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3122344"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Guzman</surname> <given-names>Hector M.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<uri xlink:href="https://loop.frontiersin.org/people/830803"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Sanchez-Galan</surname> <given-names>Javier E.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<uri xlink:href="https://loop.frontiersin.org/people/1275990"/>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Facultad de Ingenier&#x000ED;a de El&#x000E9;ctrica, Universidad Tecnol&#x000F3;gica de Panam&#x000E1;</institution>, <city>Panama City</city>, <country country="pa">Panama</country></aff>
<aff id="aff2"><label>2</label><institution>Naos Marine Laboratory, Smithsonian Tropical Research Institute</institution>, <city>Panama City</city>, <country country="pa">Panama</country></aff>
<aff id="aff3"><label>3</label><institution>Facultad de Ingenier&#x000ED;a de Sistemas Computacionales, Universidad Tecnol&#x000F3;gica de Panam&#x000E1;</institution>, <city>Panama City</city>, <country country="pa">Panama</country></aff>
<author-notes>
<corresp id="c001"><label>&#x0002A;</label>Correspondence: Javier E. Sanchez-Galan, <email xlink:href="mailto:javier.sanchezgalan@utp.ac.pa">javier.sanchezgalan@utp.ac.pa</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-08">
<day>08</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1660388</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2026 Merchan, Contreras, Poveda, Est&#x000E9;vez, Guzman and Sanchez-Galan.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Merchan, Contreras, Poveda, Est&#x000E9;vez, Guzman and Sanchez-Galan</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-08">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Demographic inference from vocalizations is essential for monitoring endangered Greater Caribbean manatees (<italic>Trichechus manatus manatus</italic>) in tropical environments where direct observation is limited. While passive acoustic monitoring has proven effective for manatee detection and individual identification, the ability to classify sex and age from vocalizations remains unexplored, limiting ecological insights into population structure and reproductive dynamics. We investigated whether machine learning can accurately classify sex and age from manatee acoustic signals using 1,285 vocalizations from 20 wild individuals captured in the Changuinola River, Panama. Acoustic features including spectral envelope descriptors (MFCCs), harmonic content (chroma), and temporal-frequency parameters were extracted and analyzed using two feature sets: SET1 (30 spectral-cepstral features) and SET2 (38 features augmented with explicit pitch and temporal descriptors). Four classification algorithms (Random Forest, XGBoost, SVM, LDA) were trained under Leave-One-Group-Out cross-validation with SMOTE oversampling to address class imbalance. Sex classification achieved 85%&#x02013;87% accuracy (75%&#x02013;78% macro-F1) with balanced performance across both classes (female: 86%, male: 79%), validating operational feasibility for passive monitoring applications. However, subject-level bootstrap analysis revealed substantial individual heterogeneity (female: 95% CI: 68.7%&#x02013;96.4%, male: 75.1%&#x02013;83.6%), indicating that approximately 10%&#x02013;15% of individuals exhibit systematic misclassification due to atypical acoustic signatures. Spectral envelope characteristics (MFCCs, spectral skewness) rather than fundamental frequency were most discriminative, suggesting sex-related variation manifests in vocal tract resonance patterns. Age classification achieved 73%&#x02013;85% global accuracy but exhibited severe juvenile under-detection (14%&#x02013;26% recall), with bootstrap confidence intervals spanning 9.3%&#x02013;86.3% for juveniles vs. 60.7%&#x02013;84.7% for adults. Dimensionality reduction (PCA, t-SNE) revealed substantial overlap between juvenile and adult acoustic feature distributions, with clearer age structure visible primarily within female clusters, contributing to systematic misclassification of male juveniles. Threshold optimization improved juvenile recall to 63% but increased false positives to 37%, presenting trade-offs for conservation surveillance. Acoustic body size regression demonstrated promising continuous estimation (MAE = 0.208 m, <italic>R</italic><sup>2</sup> &#x0003D; 0.33), offering an alternative to categorical age classification by enabling coarse demographic profiling when integrated with sex inference. These findings establish the operational viability of acoustic sex classification for manatee conservation while highlighting fundamental challenges in categorical age inference due to continuous ontogenetic variation and limited juvenile samples. However, acoustic body size regression offers a promising complementary approach, enabling continuous demographic profiling across size classes rather than discrete age categories. Integration with established individual identification frameworks would enable comprehensive acoustic mark-recapture, simultaneously estimating abundance, sex ratios, size distributions, and demographic structure from long-term hydrophone deployments without requiring visual confirmation of body dimensions.</p></abstract>
<kwd-group>
<kwd>acoustic demographic classification</kwd>
<kwd>bioacoustic classification</kwd>
<kwd>demographic inference</kwd>
<kwd>Greater Caribbean manatee</kwd>
<kwd>machine learning</kwd>
<kwd>passive acoustic monitoring (PAM)</kwd>
<kwd>vocalization analysis</kwd>
<kwd>XGBoost</kwd>
</kwd-group>
<funding-group>
<award-group id="gs1">
<funding-source id="sp1">
<institution-wrap>
<institution>Secretar&#x000ED;a Nacional de Ciencia, Tecnolog&#x000ED;a e Innovaci&#x000F3;n</institution>
<institution-id institution-id-type="doi" vocab="open-funder-registry" vocab-identifier="10.13039/open_funder_registry">10.13039/501100007061</institution-id>
</institution-wrap>
</funding-source>
<award-id rid="sp1">FID18-76</award-id>
<award-id rid="sp1"> FID21-90</award-id>
<award-id rid="sp1">FID23-106</award-id>
</award-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. This research was supported by the Secretar&#x000ED;a Nacional de Ciencia Tecnolog&#x000ED;a e Innnovacion (SENACYT-Panama) through contracts FID18-76, FID21-90, and FID23-106, and the Smithsonian Tropical Research Institute (STRI). The Sistema Nacional de Investigaci&#x000F3;n (SNI), SENACYT-Panama supports research activities by FM, HP, JS-G, and HG.</funding-statement>
</funding-group>
<counts>
<fig-count count="7"/>
<table-count count="9"/>
<equation-count count="0"/>
<ref-count count="57"/>
<page-count count="19"/>
<word-count count="11119"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Pattern Recognition</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>The Greater Caribbean manatee (<italic>Trichechus manatus manatus</italic>), a subspecies of the American manatee, is an endangered marine mammal distributed throughout the Caribbean, Gulf of Mexico, and Atlantic coast of South America (<xref ref-type="bibr" rid="B38">Morales-Vela et al., 2024</xref>). Greater Caribbean manatees face significant threats, including habitat degradation, hunting, boat collisions, and low genetic variability (<xref ref-type="bibr" rid="B28">Lefebvre et al., 2001</xref>; <xref ref-type="bibr" rid="B7">Castelblanco-Mart&#x00301;&#x00131;nez et al., 2012</xref>; <xref ref-type="bibr" rid="B22">Hines et al., 2012</xref>; <xref ref-type="bibr" rid="B11">Diaz-Ferguson et al., 2017</xref>; <xref ref-type="bibr" rid="B19">Guzman and Condit, 2017</xref>).</p>
<p>Passive acoustic monitoring has emerged as a promising tool for studying Greater Caribbean manatees, as they produce distinct vocalizations that convey information about social interactions and individual identity (<xref ref-type="bibr" rid="B50">Sousa-Lima et al., 2008</xref>; <xref ref-type="bibr" rid="B35">Merchan et al., 2019</xref>, <xref ref-type="bibr" rid="B34">2024</xref>; <xref ref-type="bibr" rid="B20">Guzman et al., 2025</xref>). Vocalizations typically include tonal calls with prominent harmonics, as well as squeaks, hi-squeaks, squeals, and chirps (<xref ref-type="bibr" rid="B4">Brady et al., 2020</xref>; <xref ref-type="bibr" rid="B3">Brady A. G. et al., 2022</xref>). Previous research has shown that manatee calls contain individually distinctive signatures and may encode cues about sex and age (<xref ref-type="bibr" rid="B54">Umeed et al., 2018</xref>; <xref ref-type="bibr" rid="B51">Sousa-Lima et al., 2002</xref>, <xref ref-type="bibr" rid="B50">2008</xref>; <xref ref-type="bibr" rid="B5">Brady E. A. et al., 2022</xref>). Studies have documented relationships between demographic traits and acoustic parameters: juveniles produce vocalizations with higher fundamental frequencies compared to adults (<xref ref-type="bibr" rid="B5">Brady E. A. et al., 2022</xref>; <xref ref-type="bibr" rid="B40">O&#x00027;Shea and Poch&#x000E9;, 2006</xref>), reflecting continuous developmental changes in vocal tract morphology, while sex-related acoustic variation appears independent of body size dimorphism&#x02014;which is minimal in <italic>Trichechus</italic> species (<xref ref-type="bibr" rid="B7">Castelblanco-Mart&#x00301;&#x00131;nez et al., 2012</xref>).</p>
<p>Machine learning frameworks for automated manatee vocal analysis have advanced rapidly. (<xref ref-type="bibr" rid="B35">Merchan et al., 2019</xref>), (<xref ref-type="bibr" rid="B36">Merchan et al., 2020</xref>), and (<xref ref-type="bibr" rid="B34">Merchan et al., 2024</xref>) established a detection-classification-clustering pipeline using CNNs and density-based clustering (HDBSCAN) for individual identification, representing one of the first large-scale frameworks for <italic>T. m. manatus</italic> monitoring. This framework was successfully deployed by (<xref ref-type="bibr" rid="B20">Guzman et al., 2025</xref>) for unsupervised individual identification of wild manatees across coastal and riverine habitats in Panama and Costa Rica, enabling estimation of residence times, site fidelity patterns, and inter-site movement dynamics from passive acoustic data alone. Complementary CNN approaches have achieved high performance in call detection (<xref ref-type="bibr" rid="B45">Rycyk et al., 2022</xref>) and vocalization type categorization (<xref ref-type="bibr" rid="B47">Schneider et al., 2024</xref>). In other taxa, machine learning has successfully extracted sex and age information from vocalizations in mice (<xref ref-type="bibr" rid="B23">Ivanenko et al., 2020</xref>), cats (<xref ref-type="bibr" rid="B53">Tavabi et al., 2021</xref>), cattle (<xref ref-type="bibr" rid="B24">Huang et al., 2021</xref>), and humans (<xref ref-type="bibr" rid="B1">Altaf and Rahman, 2023</xref>), demonstrating that acoustic signals carry biologically meaningful demographic information. However, no study has applied such methods to classify sex or age in manatees.</p>
<p>Acoustic demographic classification faces three methodological challenges. First, correlated confounding variables introduce spurious associations: body size correlates with age and influences acoustic parameters in manatees (<xref ref-type="bibr" rid="B40">O&#x00027;Shea and Poch&#x000E9;, 2006</xref>; <xref ref-type="bibr" rid="B5">Brady E. A. et al., 2022</xref>). To address age-related size confounding, statistical methods such as Analysis of Covariance (ANCOVA) can partial out the influence of body size before classification (<xref ref-type="bibr" rid="B41">Pourhoseingholi et al., 2012</xref>; <xref ref-type="bibr" rid="B16">Garc&#x000ED;a et al., 2018</xref>), ensuring that models capture genuine age-specific vocal signatures independent of allometric scaling. Second, class imbalance arising from unequal demographic representation requires techniques such as SMOTE oversampling (<xref ref-type="bibr" rid="B8">Chawla et al., 2002</xref>) combined with class-weighted loss functions. Third, individual-level generalization demands cross-validation strategies that evaluate performance across unseen individuals rather than across calls, preventing overfitting to individual-specific vocal idiosyncrasies (<xref ref-type="bibr" rid="B56">Wierucka et al., 2025</xref>).</p>
<p>Despite growing acoustic monitoring capabilities, the ability to extract reliable demographic information from Greater Caribbean manatee vocalizations remains unexplored. This capability is critical for population structure assessment, sex-ratio estimation, reproductive dynamics monitoring, and tracking vulnerable groups. In this study, we investigate whether machine learning can accurately classify sex and age from vocalizations of 20 wild manatees captured in the Changuinola River, Panama (1,285 vocalizations). We compare four supervised learning algorithms&#x02014;Random Forest (RF), Extreme Gradient Boosting (XGBoost), Support Vector Machine (SVM), and Linear Discriminant Analysis (LDA)&#x02014;under Leave-One-Group-Out cross-validation, employing ANCOVA residualization to control size confounding and SMOTE oversampling to address class imbalance. We assess classification performance through bootstrap-derived confidence intervals to quantify individual-level heterogeneity, evaluate threshold optimization strategies for minority class detection, and explore body size estimation as an alternative continuous demographic proxy. This comparative framework identifies optimal approaches for acoustic demographic inference in passive monitoring contexts, providing uncertainty metrics essential for evidence-based conservation decision-making.</p></sec>
<sec sec-type="materials|methods" id="s2">
<label>2</label>
<title>Materials and methods</title>
<sec>
<label>2.1</label>
<title>Vocalization data set</title>
<p>The data used in the experiments of this manuscript come from a data set previously presented in an article by (<xref ref-type="bibr" rid="B34">Merchan et al., 2024</xref>). Individual manatees were captured using a custom-designed 4 &#x000D7; 4 m floating enclosure made from 20 cm diameter HDPE pipes, which supported a fishing net with an 8 cm mesh and a depth of 2.5 m (see <xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Floating cage where manatees were temporarily captured for recording (San San River, Bocas del Toro, Panama).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1660388-g0001.tif">
<alt-text content-type="machine-generated">A floating black frame structure on a calm river, surrounded by dense green trees reflecting on the water. The blue sky is partially visible through the foliage.</alt-text>
</graphic>
</fig>
<p>This structure was anchored with ropes tied to nearby trees and positioned in the center of a channel 40 m wide and 3&#x02013;5 m deep in the upstream section of the San San River, Bocas del Toro, Panama (coordinates: 09&#x000B0;.979&#x02032; N; 82&#x000B0;32.964&#x02032; W). To attract manatees into the enclosure without feeding them, a wire was suspended with a bucket filled with fresh banana pulp and banana leaves. Entry was manually operated from the riverbank side through a stainless-steel gate measuring 1.5 &#x000D7; 1.8 m facing the deeper side of the river. After entry, the gate was closed and the manatees remained inside for 6 to 8 h.</p>
<p>During confinement, vocalizations were recorded using a micro-RUDAR<sup>&#x000AE;</sup>system (Cetacean Research, Seattle, Washington) equipped with an SQ26-08 hydrophone connected to an H1 Zoom<sup>&#x000AE;</sup>digital recorder, set for continuous recording at 96 kHz and 24-bit resolution for 6&#x02013;10 h. Each animal was measured (&#x000B1;10 cm accuracy) using a tape measure, with the floating structure serving as a reference scale. The sex of each manatee was visually determined while the animal swam and rotated inside the cage, based on external anatomical traits: in males, the genital slit is positioned closer to the umbilicus and no mammary glands are present, whereas in females the genital opening is located nearer to the anus and is flanked by mammary glands under the flippers (<xref ref-type="bibr" rid="B21">Hartman, 1979</xref>; <xref ref-type="bibr" rid="B43">Reynolds and Odell, 1992</xref>).</p>
<p>Age class (juvenile or adult) was assigned based on body length and external morphological indicators of sexual maturity assessed during capture. While definitive maturity determination requires histological or hormonal analysis (<xref ref-type="bibr" rid="B30">Marmontel, 1995</xref>; <xref ref-type="bibr" rid="B7">Castelblanco-Mart&#x00301;&#x00131;nez et al., 2012</xref>), these field-based assessments provide operational demographic categories suitable for acoustic classification studies. Maturity assessment prioritized morphological indicators (genital development, body proportions, scarring patterns) over absolute body size, as sexual maturity in <italic>Trichechus manatus</italic> exhibits individual variation and does not follow a strict size threshold. Published maturity thresholds are primarily available for Florida manatees (<italic>T. m. latirostris</italic>: 2.1&#x02013;2.5 m; <xref ref-type="bibr" rid="B30">Marmontel 1995</xref>; <xref ref-type="bibr" rid="B40">O&#x00027;Shea and Poch&#x000E9; 2006</xref>), which may differ from Greater Caribbean populations. Field observations for this study yielded an approximate empirical threshold of &#x0007E;2.2 m, though individual variation resulted in overlap between size ranges of juvenile and adult individuals (juveniles: 1.70&#x02013;2.20 m; adults: 2.20&#x02013;3.00 m), reflecting the continuous nature of ontogenetic development.</p>
<p>Photographs of scars or identification marks on the face and body were also taken for future individual identification. After 6&#x02013;8 h, the manatees were released. All procedures were carried out with the approval of the Animal Care and Use Committee of the Smithsonian Tropical Research Institute (IACUC).</p>
<p>As in (<xref ref-type="bibr" rid="B34">Merchan et al., 2024</xref>), the methodology involves detecting, extracting, and confirming all manatee vocalizations. The dataset was constructed by first isolating the vocalizations from the continuous acoustic recordings and then analyzing them using the detection framework introduced in (<xref ref-type="bibr" rid="B35">Merchan et al., 2019</xref>) and (<xref ref-type="bibr" rid="B36">Merchan et al., 2020</xref>). This procedure consisted of three main steps: (i) a detection phase based on the analysis of the Autocorrelation Function (ACF) (<xref ref-type="bibr" rid="B35">Merchan et al., 2019</xref>), (ii) a denoising phase to enhance signal quality, and (iii) a classification phase using a CNN (<xref ref-type="bibr" rid="B36">Merchan et al., 2020</xref>). During the classification stage, the candidate signals identified in the detection phase were evaluated by a pre-trained CNN model capable of distinguishing manatee vocalizations from environmental sounds, thereby validating the detections and producing a dataset of confirmed calls.</p>
<p>Twenty-eight manatees were captured in total; however, three were recaptures and were therefore excluded from the analysis. In addition, for five individuals, it was not possible to observe the anatomical features necessary to determine sex due to low illumination during nocturnal capture and therefore, were not included in the final data set. Using the described methodology, a data set consisting of 1,285 vocalizations from 20 unique individuals, along with their corresponding characteristics (estimated age and sex), was generated. The overall characteristics of the dataset are described in <xref ref-type="table" rid="T1">Table 1</xref>. Most vocalizations in this dataset are classified as squeaks or high-squeaks; however, for four individuals (S07, S18, S23, and S27), a substantial proportion of their vocalizations were identified as squeals with 33%, 82%, 63%, and 80%, respectively.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Catalog of manatees temporarily captured in San San River (20 individuals with known sex).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>ID</bold></th>
<th valign="top" align="center"><bold>Subject</bold></th>
<th valign="top" align="center"><bold>Capture date</bold></th>
<th valign="top" align="center"><bold>Sex</bold></th>
<th valign="top" align="center"><bold>Age</bold></th>
<th valign="top" align="center"><bold>Size (m)</bold></th>
<th valign="top" align="center"><bold>&#x00023; Voc</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">M01</td>
<td valign="top" align="center">S03</td>
<td valign="top" align="center">22-Jan-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.50</td>
<td valign="top" align="center">55</td>
</tr>
<tr>
<td valign="top" align="left">M02</td>
<td valign="top" align="center">S04</td>
<td valign="top" align="center">24-Jan-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.70</td>
<td valign="top" align="center">31</td>
</tr>
<tr>
<td valign="top" align="left">M03</td>
<td valign="top" align="center">S07</td>
<td valign="top" align="center">21-Apr-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.90</td>
<td valign="top" align="center">54</td>
</tr>
<tr>
<td valign="top" align="left">M04</td>
<td valign="top" align="center">S09</td>
<td valign="top" align="center">19-May-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.50</td>
<td valign="top" align="center">54</td>
</tr>
<tr>
<td valign="top" align="left">M05</td>
<td valign="top" align="center">S10</td>
<td valign="top" align="center">21-May-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.70</td>
<td valign="top" align="center">63</td>
</tr>
<tr>
<td valign="top" align="left">M06</td>
<td valign="top" align="center">S12</td>
<td valign="top" align="center">04-Jul-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.90</td>
<td valign="top" align="center">61</td>
</tr>
<tr>
<td valign="top" align="left">M07</td>
<td valign="top" align="center">S13</td>
<td valign="top" align="center">06-Jul-2021</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">2.20</td>
<td valign="top" align="center">78</td>
</tr>
<tr>
<td valign="top" align="left">M08</td>
<td valign="top" align="center">S14</td>
<td valign="top" align="center">23-Aug-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.30</td>
<td valign="top" align="center">70</td>
</tr>
<tr>
<td valign="top" align="left">M09</td>
<td valign="top" align="center">S15</td>
<td valign="top" align="center">23-Oct-2021</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.20</td>
<td valign="top" align="center">88</td>
</tr>
<tr>
<td valign="top" align="left">M10</td>
<td valign="top" align="center">S16</td>
<td valign="top" align="center">24-Oct-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.80</td>
<td valign="top" align="center">76</td>
</tr>
<tr>
<td valign="top" align="left">M11</td>
<td valign="top" align="center">S17</td>
<td valign="top" align="center">25-Oct-2021</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.80</td>
<td valign="top" align="center">60</td>
</tr>
<tr>
<td valign="top" align="left">M12</td>
<td valign="top" align="center">S18</td>
<td valign="top" align="center">26-Oct-2021</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">1.70</td>
<td valign="top" align="center">72</td>
</tr>
<tr>
<td valign="top" align="left">M13</td>
<td valign="top" align="center">S19</td>
<td valign="top" align="center">09-Mar-2022</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.80</td>
<td valign="top" align="center">52</td>
</tr>
<tr>
<td valign="top" align="left">M14</td>
<td valign="top" align="center">S21</td>
<td valign="top" align="center">19-Jun-2022</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.50</td>
<td valign="top" align="center">68</td>
</tr>
<tr>
<td valign="top" align="left">M15</td>
<td valign="top" align="center">S22</td>
<td valign="top" align="center">20-Jun-2022</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">1.80</td>
<td valign="top" align="center">63</td>
</tr>
<tr>
<td valign="top" align="left">M16</td>
<td valign="top" align="center">S23</td>
<td valign="top" align="center">21-Jun-2022</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.80</td>
<td valign="top" align="center">63</td>
</tr>
<tr>
<td valign="top" align="left">M17</td>
<td valign="top" align="center">S24</td>
<td valign="top" align="center">08-Aug-2022</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">2.10</td>
<td valign="top" align="center">57</td>
</tr>
<tr>
<td valign="top" align="left">M18</td>
<td valign="top" align="center">S25</td>
<td valign="top" align="center">09-Aug-2022</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">3.00</td>
<td valign="top" align="center">54</td>
</tr>
<tr>
<td valign="top" align="left">M19</td>
<td valign="top" align="center">S27</td>
<td valign="top" align="center">09-Jan-2023</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">2.35</td>
<td valign="top" align="center">78</td>
</tr>
<tr>
<td valign="top" align="left">M20</td>
<td valign="top" align="center">S28</td>
<td valign="top" align="center">18-May-2023</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">1.95</td>
<td valign="top" align="center">88</td>
</tr>
<tr>
<td/>
<td/>
<td/>
<td/>
<td/>
<td valign="top" align="center"><bold>Total</bold></td>
<td valign="top" align="center"><bold>1,285</bold></td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<label>2.2</label>
<title>Signal preprocessing</title>
<p>Passive Acoustic Monitoring (PAM) devices typically operate at sampling rates of 44.1&#x02013;96 kHz. Raw recordings (96 kHz) were downsampled to 48 kHz to emulate realistic field conditions.</p>
<p>Riverine and coastal environments exhibit high low-frequency ambient noise (0&#x02013;1 kHz) from boat engines, water flow, and fish vocalizations (<xref ref-type="bibr" rid="B13">Erbe et al., 2013</xref>; <xref ref-type="bibr" rid="B37">Miksis-Olds et al., 2018</xref>). An 8th-order Butterworth high-pass filter (cutoff: 1 kHz) was applied to attenuate environmental interference while preserving manatee vocalizations (2&#x02013;24 kHz range, with <italic>f</italic><sub>0</sub> concentrated at 3&#x02013;8 kHz depending on age) (<xref ref-type="bibr" rid="B5">Brady E. A. et al., 2022</xref>; <xref ref-type="bibr" rid="B50">Sousa-Lima et al., 2008</xref>).</p>
</sec>
<sec>
<label>2.3</label>
<title>Acoustic feature extraction</title>
<p>Two hierarchical feature configurations were evaluated. Feature SET1 (30 dimensions) comprised: 12 Mel-Frequency Cepstral Coefficients (MFCCs) (<xref ref-type="bibr" rid="B29">Logan, 2000</xref>), 12 chroma pitch class profiles (<xref ref-type="bibr" rid="B39">M&#x000FC;ller and Ewert, 2007</xref>), and six spectral shape statistics (centroid, flatness, flux, skewness, kurtosis, entropy). Feature SET2 (38 dimensions) augmented SET1 with eight additional parameters: six temporal-frequency features and two harmonic structure descriptors.</p>
<p>MFCCs were computed using librosa (<xref ref-type="bibr" rid="B32">McFee et al., 2015</xref>) with default parameters, retaining 12 coefficients and omitting the 0th coefficient. Chroma features were extracted using Short-Time Fourier Transform (STFT) with FFT window size of 2,048 samples and hop length of 1,536 samples. Spectral statistics (centroid, flatness, flux, skewness, kurtosis, entropy) were derived from magnitude spectra, computed frame-wise and averaged across call duration.</p>
<p>Considering the acoustic features employed by <xref ref-type="bibr" rid="B50">Sousa-Lima et al. (2008</xref>) for Antillean manatee vocal analysis, we extracted temporal-frequency and harmonic parameters. Fundamental frequency was estimated using the probabilistic YIN (PYIN) algorithm (<xref ref-type="bibr" rid="B31">Mauch and Dixon, 2014</xref>) with parameters optimized for Greater Caribbean manatee vocal range: fmin = 1,000 Hz and fmax = 8,000 Hz, consistent with the reported range of 1.07&#x02013;4.98 kHz (<xref ref-type="bibr" rid="B51">Sousa-Lima et al., 2002</xref>). The <italic>f</italic><sub>0</sub> estimate corresponded to the pitch value with maximum voicing probability across frames. Following <xref ref-type="bibr" rid="B50">Sousa-Lima et al. (2008</xref>), we extracted maximum <italic>f</italic><sub>0</sub>, minimum <italic>f</italic><sub>0</sub>, and peak frequency (frequency with maximum energy in the FFT magnitude spectrum), as well as call duration and frequency modulation (computed as the range of detected <italic>f</italic><sub>0</sub> values). Additionally, we characterized harmonic structure by detecting spectral peaks at integer multiples of <italic>f</italic><sub>0</sub> within a &#x000B1;50 Hz tolerance window and &#x02013;40 dB threshold relative to the maximum peak, yielding the number of harmonics (range: 1&#x02013;19, mean: 9.6) and the harmonic with maximum energy. All features were standardized to zero mean and unit variance (<xref ref-type="bibr" rid="B27">Kuhn and Johnson, 2013</xref>).</p>
</sec>
<sec>
<label>2.4</label>
<title>Confounding variable control via ANCOVA residualization</title>
<p>Body size correlates strongly with age, potentially confounding age classification (<xref ref-type="bibr" rid="B21">Hartman, 1979</xref>). To isolate demographic-specific acoustic signatures, we applied ANCOVA residualization to remove size-related variance from features before classification (<xref ref-type="bibr" rid="B48">Searle et al., 2017</xref>).</p>
<p>For age classification, features were regressed against body size (continuous covariate) controlling for sex. For sex classification, features were regressed against body size controlling for age. Critically, to prevent data leakage, ANCOVA residualization was performed independently within each cross-validation fold using the following procedure:</p>
<list list-type="order">
<list-item><p>For each LOGO fold, the linear regression model (including intercept, body size coefficient, and demographic covariates) was fitted exclusively on that fold&#x00027;s training set.</p></list-item>
<list-item><p>Regression coefficients and residual variance estimates were computed solely from the training data.</p></list-item>
<list-item><p>The fitted model parameters were then applied to both the training set and the held-out test set to compute residuals, ensuring that no information from test samples influenced the model fitting process.</p></list-item>
<list-item><p>Residuals from these linear models replaced raw features. For age classification, this preserved age-related variation while removing size-related confounding. For sex classification, this preserved sex-related variation while removing size-related confounding.</p></list-item>
</list>
<p>Two regularization strategies were evaluated: a fixed parameter &#x003C4; &#x0003D; 1.2 and an adaptive parameter <inline-formula><mml:math id="M1"><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:msqrt><mml:mrow><mml:mi>p</mml:mi><mml:mo>/</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msqrt></mml:math></inline-formula>, where <italic>p</italic> is the number of features and <italic>n</italic> is the training sample size per fold. The fixed parameter (&#x003C4; &#x0003D; 1.2) provides consistent regularization across all folds, retaining 85%&#x02013;92% of original variance. The adaptive approach dynamically adjusts regularization strength based on dimensionality and sample size within each fold. The regularization parameter &#x003C4; was determined from training set characteristics and applied consistently to both training and test transformations within each fold, preventing information leakage. Both strategies were applied during the feature selection step, with residualized features serving as input to subsequent classification stages.</p>
</sec>
<sec>
<label>2.5</label>
<title>Feature selection</title>
<p>Following ANCOVA residualization, two feature selection strategies were compared: univariate filtering (SelectKBest) and Recursive Feature Elimination (RFE).</p>
<p>SelectKBest retained the top <italic>k</italic> features based on ANOVA F-statistics (<xref ref-type="bibr" rid="B17">Guyon and Elisseeff, 2003</xref>), with <italic>k</italic> &#x0003D; 20 for SET1 (from 30 features) and <italic>k</italic> &#x0003D; 28 for SET2 (from 38 features), preserving approximately 67%&#x02013;74% of features. This approach evaluates features independently without considering classifier interactions.</p>
<p>Recursive Feature Elimination (RFE) (<xref ref-type="bibr" rid="B18">Guyon et al., 2002</xref>) provided a model-aware alternative by iteratively training XGBoost classifiers, ranking features by importance, and eliminating the least informative feature until optimal subset size was determined via three-fold cross-validation (minimum five features). Unlike SelectKBest, RFE accounts for feature dependencies and classifier-specific relevance, potentially identifying more discriminative subsets at the cost of increased computational expense.</p>
<p>Both strategies were evaluated within the cross-validation pipeline: feature selection was performed independently on each training fold to prevent information leakage, and the selected subset was then applied to the corresponding test fold. This ensures unbiased performance estimates and fair comparison between univariate and multivariate selection approaches.</p>
</sec>
<sec>
<label>2.6</label>
<title>Class imbalance mitigation via stratified SMOTE</title>
<p>Class imbalances existed for both demographic dimensions (female:male = 2.3:1; adult:juvenile = 2.5:1). To prevent systematic bias toward majority classes, we applied Synthetic Minority Over-sampling Technique (SMOTE) (<xref ref-type="bibr" rid="B8">Chawla et al., 2002</xref>) within each training fold prior to model fitting. SMOTE generates synthetic minority-class examples by interpolating between nearest neighbors in feature space, creating linearly interpolated instances along the line segments connecting minority samples. The number of neighbors was dynamically adjusted based on sample availability (<italic>k</italic> &#x0003D; 5 when sufficient minority samples existed, otherwise reduced to <italic>k</italic> &#x0003D; 1 to accommodate sparse demographic combinations).</p>
<p>Critically, to address class imbalance while avoiding the introduction of spurious age-sex associations arising from skewed demographic distributions among individuals (adults are 91% female, 9% male; juveniles are 16% female, 84% male), SMOTE was applied with stratification by the auxiliary demographic variable. Specifically, when training the sex classifier, SMOTE balanced female and male classes while maintaining proportional representation of age groups within each sex class. Conversely, when training the age classifier, SMOTE balanced adult and juvenile classes while preserving proportional representation of sex within each age class. This stratified approach ensures that synthetic samples respect the joint demographic distribution within each target class, reducing the risk that classifiers exploit spurious demographic cues introduced by oversampling.</p>
<p>SMOTE was applied exclusively to training folds; test folds retained their natural class distributions to preserve ecological validity. Oversampling achieved approximate 1:1 target class ratios in training data, though stratification constraints and singleton handling prevented perfect balance when auxiliary class distributions were highly skewed.</p>
</sec>
<sec>
<label>2.7</label>
<title>Classification algorithms</title>
<p>Four supervised learning algorithms were compared:</p>
<p>XGBoost: gradient-boosted decision trees (<xref ref-type="bibr" rid="B9">Chen and Guestrin, 2016</xref>) with 100 estimators, learning rate 0.1, maximum depth 3, minimum child weight 2, subsample ratio 0.8, and column subsample ratio 0.8. Scale positive weight parameter adjusted based on class imbalance to handle residual imbalance post-SMOTE. Single-thread execution ensured reproducibility.</p>
<p>Random Forest (RF): ensemble of 100 decision trees (<xref ref-type="bibr" rid="B6">Breiman, 2001</xref>) with maximum depth 5, minimum samples per split 10, minimum samples per leaf 4, and <inline-formula><mml:math id="M2"><mml:msqrt><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msqrt></mml:math></inline-formula> features per split. Class weights set to &#x0201C;balanced&#x0201D; to handle residual imbalance post-SMOTE.</p>
<p>Support vector machines (SVM): radial basis function kernel (<xref ref-type="bibr" rid="B10">Cortes and Vapnik, 1995</xref>) with &#x003B3;= &#x0201C;scale&#x0201D; and regularization <italic>C</italic> &#x0003D; 1.0. Class weights set to &#x0201C;balanced&#x0201D; with probability estimates enabled for threshold optimization experiments.</p>
<p>Linear discriminant analysis (LDA): default solver with automatic shrinkage estimation (<xref ref-type="bibr" rid="B14">Fisher, 1936</xref>).</p>
<p>All hyperparameters were fixed a priori to ensure consistency across experimental conditions. Random state set to 42 for reproducibility.</p>
</sec>
<sec>
<label>2.8</label>
<title>Cross-validation strategy</title>
<p>We employed Leave-One-Group-Out (LOGO) cross-validation with individuals as groups (<xref ref-type="bibr" rid="B2">Arlot and Celisse, 2010</xref>). Each fold withholds all calls from a single subject as the test set, training on the remaining individuals. This procedure iterates across all 20 subjects (complete dataset) or 16 subjects (squeal-reduced dataset).</p>
<p>LOGO ensures generalization to unseen individuals, critical for field deployment where models must classify vocalizations from previously unencountered animals (<xref ref-type="bibr" rid="B55">Varma and Simon, 2006</xref>). Mean accuracy and per-class recall across folds quantify expected performance on novel individuals.</p>
</sec>
<sec>
<label>2.9</label>
<title>Performance evaluation</title>
<p>Classification performance was evaluated via accuracy (fraction of correct predictions), per-class recall (true positive rate), precision (positive predictive value), and F1-score (<xref ref-type="bibr" rid="B49">Sokolova and Lapalme, 2009</xref>). For imbalanced datasets, per-class metrics provide insight into minority detection capability.</p>
<p>To assess model performance independent of class prevalence, we computed macro-averaged metrics (unweighted mean across classes) alongside standard weighted accuracy. Macro-average precision, recall, and F1-score provide equal weight to minority classes, revealing whether models achieve balanced performance or systematically favor dominant demographics (<xref ref-type="bibr" rid="B49">Sokolova and Lapalme, 2009</xref>).</p>
<p>All metrics were computed per fold and aggregated via mean &#x000B1; standard deviation (SD) across folds. To quantify subject-level variability and address potential heterogeneity in acoustic signatures, we additionally computed 95% confidence intervals (CI) via bootstrap resampling over individuals (<xref ref-type="bibr" rid="B12">Efron and Tibshirani, 1994</xref>). For each demographic class, individual-level accuracy was calculated from all vocalizations per subject, then bootstrap resampling (<italic>n</italic> &#x0003D; 10, 000 iterations) was applied at the subject level (not vocalization level) to estimate variability in mean accuracy. This subject-based bootstrap accounts for within-individual correlation and provides unbiased estimates of performance uncertainty across the population.</p>
<p>Threshold optimization experiments for both sex and age classification evaluated recall-precision trade-offs by varying decision thresholds from 0.05 to 0.95 in 0.05 increments (<xref ref-type="bibr" rid="B42">Provost and Fawcett, 2001</xref>). For sex classification, we monitored male-class metrics (recall, precision, F1-score, and subject-level SD) to identify the threshold maximizing male F1-score. For age classification, we monitored juvenile-class metrics to optimize detection of the minority age class. This analysis quantifies the trade-off between minority-class sensitivity and overall classification accuracy.</p>
<p>Feature importance was assessed via mean decrease in impurity for tree-based methods (Random Forest, XGBoost) (<xref ref-type="bibr" rid="B6">Breiman, 2001</xref>), averaged across LOGO folds and normalized to sum to 1.0. For each fold, feature importance was extracted from the trained classifier and aggregated to identify consistently discriminative acoustic parameters.</p>
</sec>
<sec>
<label>2.10</label>
<title>Body size estimation via acoustic regression</title>
<p>To evaluate non-invasive body size estimation from vocalizations, we trained Random Forest and XGBoost regressors (200 estimators each) to predict body length from acoustic features alone under LOGO cross-validation. Both feature sets (SET1: 30 spectral-cepstral features; SET2: 38 features augmented with temporal-frequency and harmonic descriptors) were tested to assess whether explicit pitch parameters improve size estimation beyond spectral envelope information. An ensemble model averaged Random Forest and XGBoost predictions. Performance was quantified via mean absolute error (MAE), root mean squared error (RMSE), and coefficient of determination (<italic>R</italic><sup>2</sup>), computed both at the vocalization level and aggregated per subject to assess individual-level prediction accuracy. This approach tests whether body length&#x02014;which correlates with vocal tract dimensions&#x02014;can be inferred directly from acoustic signatures without prior knowledge of sex or age categories.</p>
</sec>
<sec>
<label>2.11</label>
<title>Software and hardware</title>
<p>All experiments were performed using Python 3.10 on a custom workstation with AMD Ryzen 9 5950X CPU, NVIDIA RTX 3080 GPU (10GB VRAM), and 128GB RAM running Ubuntu 22.04. Python libraries included NumPy (1.24.0), Scikit-learn (1.3.0), Librosa (0.10.0), XGBoost (1.7.6), imbalanced-learn (0.11.0), and SciPy (1.11.0).</p>
</sec>
<sec>
<label>2.12</label>
<title>Experimental design</title>
<p>We evaluated four supervised learning algorithms (XGBoost, Random Forest, SVM, Linear Discriminant Analysis) across two feature configurations, two dataset variants, and two feature selection strategies for demographic classification. SET1 comprised 30 spectral-cepstral features (12 MFCCs, 12 chroma pitch classes, six spectral shape descriptors); SET2 added eight temporal-frequency and harmonic parameters (call duration, mean/max/min <italic>f</italic><sub>0</sub>, peak frequency, frequency modulation, number of harmonics, harmonic with maximum energy). This assessed whether explicit pitch features complement spectral envelope information from MFCCs.</p>
<p>Feature selection compared univariate filtering (SelectKBest: <italic>k</italic> &#x0003D; 20 for SET1, <italic>k</italic> &#x0003D; 28 for SET2 via ANOVA F-statistics) vs. Recursive Feature Elimination (RFE: cross-validated optimization, minimum five features). RFE captures feature interactions and classifier-specific relevance at higher computational cost.</p>
<p>Our complete dataset (1,285 vocalizations, 20 subjects) was compared to a squeal-reduced variant (1,018 vocalizations, 16 subjects) excluding four individuals whose repertoires consisted predominantly of acoustically distinct high-frequency squeals. Leave-One-Group-Out cross-validation ensured generalization to unseen individuals for both classification and regression tasks.</p>
<p>For demographic classification, the design yielded 64 configurations per task: four classifiers &#x000D7; two feature sets &#x000D7; two datasets &#x000D7; two selection methods &#x000D7; two ANCOVA regularization strategies (fixed &#x003C4; &#x0003D; 1.2 vs. adaptive <inline-formula><mml:math id="M3"><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msqrt><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>/</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>). Within each fold: ANCOVA residualization removed size confounding, feature selection reduced dimensionality, SMOTE with auxiliary demographic stratification addressed class imbalances (female:male = 2.3:1; adult:juvenile = 2.5:1), and classifiers trained with class-weight balancing. For body size estimation, Random Forest and XGBoost regressors were trained on both feature sets with acoustic features alone or augmented with demographic predictors, evaluated via MAE, RMSE, and <italic>R</italic><sup>2</sup>.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<label>3</label>
<title>Results</title>
<sec>
<label>3.1</label>
<title>Overall classification performance</title>
<p>Classification performance was evaluated across 64 experimental configurations per demographic task, varying classifier architecture, feature dimensionality, dataset composition, and feature selection strategy. <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref> present comprehensive performance metrics across LOGO folds for the four supervised learning algorithms under univariate feature selection (SelectKBest), including accuracy, macro-F1, and per-class precision, recall, and F1-score. Subject-level performance and bootstrap-derived 95% confidence intervals are detailed in <xref ref-type="table" rid="T4">Tables 4</xref>, <xref ref-type="table" rid="T5">5</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Sex classification performance across methods, feature sets, and datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="center"><bold>Features</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Acc (%)</bold></th>
<th valign="top" align="center"><bold>Macro-F1</bold></th>
<th valign="top" align="center" colspan="3"><bold>Female</bold></th>
<th valign="top" align="center" colspan="3"><bold>Male</bold></th>
</tr>
 <tr>
<th/>
<th/>
<th/>
<th/>
<th/>
<th valign="top" align="center"><bold>Prec</bold></th>
<th valign="top" align="center"><bold>Rec</bold></th>
<th valign="top" align="center"><bold>F1</bold></th>
<th valign="top" align="center"><bold>Prec</bold></th>
<th valign="top" align="center"><bold>Rec</bold></th>
<th valign="top" align="center"><bold>F1</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Random forest</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">84.6</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.75</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">83.6</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.73</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">84.8</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">0.75</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">84.2</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.74</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">85.1</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.77</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">84.5</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.75</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center"><bold>87.3</bold></td>
<td valign="top" align="center"><bold>0.80</bold></td>
<td valign="top" align="center"><bold>0.89</bold></td>
<td valign="top" align="center"><bold>0.91</bold></td>
<td valign="top" align="center"><bold>0.90</bold></td>
<td valign="top" align="center"><bold>0.82</bold></td>
<td valign="top" align="center"><bold>0.77</bold></td>
<td valign="top" align="center"><bold>0.79</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">86.8</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.77</td>
</tr>
<tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">82.4</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.71</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">81.8</td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.65</td>
<td valign="top" align="center">0.69</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">83.7</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.73</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">83.2</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.71</td>
</tr>
<tr>
<td valign="top" align="left">LDA</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">79.5</td>
<td valign="top" align="center">0.69</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.69</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.65</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">78.9</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.63</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">80.8</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.64</td>
<td valign="top" align="center">0.67</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">80.3</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.65</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>All models trained via Leave-One-Group-Out cross-validation with SMOTE oversampling. Metrics: accuracy (Acc), Macro-F1, per-class precision (Prec), recall (Rec), and F1-score. Best performance in bold.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Age classification performance across methods, feature sets, and datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="center"><bold>Features</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Acc (%)</bold></th>
<th valign="top" align="center"><bold>Macro-F1</bold></th>
<th valign="top" align="center" colspan="3"><bold>Adult</bold></th>
<th valign="top" align="center" colspan="3"><bold>Juvenile</bold></th>
</tr>
<tr>
<th/>
<th/>
<th/>
<th/>
<th/>
<th valign="top" align="center"><bold>Prec</bold></th>
<th valign="top" align="center"><bold>Rec</bold></th>
<th valign="top" align="center"><bold>F1</bold></th>
<th valign="top" align="center"><bold>Prec</bold></th>
<th valign="top" align="center"><bold>Rec</bold></th>
<th valign="top" align="center"><bold>F1</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Random forest</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">71.8</td>
<td valign="top" align="center">0.54</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.14</td>
<td valign="top" align="center">0.23</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">70.4</td>
<td valign="top" align="center">0.52</td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.11</td>
<td valign="top" align="center">0.19</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">74.2</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.22</td>
<td valign="top" align="center">0.33</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">73.0</td>
<td valign="top" align="center">0.56</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.95</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.64</td>
<td valign="top" align="center">0.17</td>
<td valign="top" align="center">0.27</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">84.2</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.26</td>
<td valign="top" align="center">0.40</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">82.7</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.23</td>
<td valign="top" align="center">0.36</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center"><bold>85.8</bold></td>
<td valign="top" align="center"><bold>0.73</bold></td>
<td valign="top" align="center"><bold>0.86</bold></td>
<td valign="top" align="center"><bold>0.98</bold></td>
<td valign="top" align="center"><bold>0.92</bold></td>
<td valign="top" align="center"><bold>0.85</bold></td>
<td valign="top" align="center"><bold>0.26</bold></td>
<td valign="top" align="center"><bold>0.40</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">84.5</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.24</td>
<td valign="top" align="center">0.37</td>
</tr>
<tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">68.5</td>
<td valign="top" align="center">0.49</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.52</td>
<td valign="top" align="center">0.09</td>
<td valign="top" align="center">0.15</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">67.2</td>
<td valign="top" align="center">0.47</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.49</td>
<td valign="top" align="center">0.06</td>
<td valign="top" align="center">0.11</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">71.3</td>
<td valign="top" align="center">0.54</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.59</td>
<td valign="top" align="center">0.15</td>
<td valign="top" align="center">0.24</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">69.8</td>
<td valign="top" align="center">0.51</td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.55</td>
<td valign="top" align="center">0.11</td>
<td valign="top" align="center">0.18</td>
</tr>
<tr>
<td valign="top" align="left">LDA</td>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">65.1</td>
<td valign="top" align="center">0.45</td>
<td valign="top" align="center">0.69</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.45</td>
<td valign="top" align="center">0.07</td>
<td valign="top" align="center">0.12</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET1</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">63.8</td>
<td valign="top" align="center">0.43</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.42</td>
<td valign="top" align="center">0.04</td>
<td valign="top" align="center">0.08</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Complete</td>
<td valign="top" align="center">68.4</td>
<td valign="top" align="center">0.51</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.52</td>
<td valign="top" align="center">0.13</td>
<td valign="top" align="center">0.21</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">SET2</td>
<td valign="top" align="left">Squeal-reduced</td>
<td valign="top" align="center">66.9</td>
<td valign="top" align="center">0.48</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.48</td>
<td valign="top" align="center">0.09</td>
<td valign="top" align="center">0.15</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>All models trained via Leave-One-Group-Out cross-validation with SMOTE oversampling. Metrics: accuracy (Acc), macro-F1, per-class precision (Prec), recall (Rec), and F1-score. Best performance in bold.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Subject-level classification performance with SD and individual 95% confidence intervals.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Subject</bold></th>
<th valign="top" align="center"><bold>Sex</bold></th>
<th valign="top" align="center"><bold>Age</bold></th>
<th valign="top" align="center"><bold><italic>N</italic> calls</bold></th>
<th valign="top" align="center" colspan="3"><bold>Sex classification</bold></th>
<th valign="top" align="center" colspan="3"><bold>Age classification</bold></th>
</tr>
 <tr>
<th/>
<th/>
<th/>
<th/>
<th valign="top" align="center"><bold>Acc (%)</bold></th>
<th valign="top" align="center"><bold>SD</bold></th>
<th valign="top" align="center"><bold>95% CI</bold></th>
<th valign="top" align="center"><bold>Acc (%)</bold></th>
<th valign="top" align="center"><bold>SD</bold></th>
<th valign="top" align="center"><bold>95% CI</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">S03</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">54</td>
<td valign="top" align="center">100.0</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">(100.0, 100.0)</td>
<td valign="top" align="center">100.0</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">(100.0, 100.0)</td>
</tr>
<tr>
<td valign="top" align="left">S04</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">31</td>
<td valign="top" align="center">100.0</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">(100.0, 100.0)</td>
<td valign="top" align="center">96.8</td>
<td valign="top" align="center">5.7</td>
<td valign="top" align="center">(87.1, 100.0)</td>
</tr>
<tr>
<td valign="top" align="left">S09</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">54</td>
<td valign="top" align="center">79.6</td>
<td valign="top" align="center">5.1</td>
<td valign="top" align="center">(68.5, 88.9)</td>
<td valign="top" align="center">72.2</td>
<td valign="top" align="center">6.7</td>
<td valign="top" align="center">(59.3, 83.3)</td>
</tr>
<tr>
<td valign="top" align="left">S10</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">63</td>
<td valign="top" align="center">96.8</td>
<td valign="top" align="center">3.9</td>
<td valign="top" align="center">(90.5, 100.0)</td>
<td valign="top" align="center">98.4</td>
<td valign="top" align="center">2.8</td>
<td valign="top" align="center">(93.7, 100.0)</td>
</tr>
<tr>
<td valign="top" align="left">S12</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">61</td>
<td valign="top" align="center">85.2</td>
<td valign="top" align="center">5.1</td>
<td valign="top" align="center">(75.4, 93.4)</td>
<td valign="top" align="center">63.9</td>
<td valign="top" align="center">6.5</td>
<td valign="top" align="center">(50.8, 75.4)</td>
</tr>
<tr>
<td valign="top" align="left">S13</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">39</td>
<td valign="top" align="center">82.1</td>
<td valign="top" align="center">7.1</td>
<td valign="top" align="center">(69.2, 92.3)</td>
<td valign="top" align="center">6.4</td>
<td valign="top" align="center">4.7</td>
<td valign="top" align="center">(0.0, 15.4)</td>
</tr>
<tr>
<td valign="top" align="left">S14</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">35</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">(0.0, 0.0)</td>
<td valign="top" align="center">98.6</td>
<td valign="top" align="center">2.2</td>
<td valign="top" align="center">(94.3, 100.0)</td>
</tr>
<tr>
<td valign="top" align="left">S15</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">72.7</td>
<td valign="top" align="center">11.0</td>
<td valign="top" align="center">(50.0, 90.9)</td>
<td valign="top" align="center">30.7</td>
<td valign="top" align="center">9.1</td>
<td valign="top" align="center">(13.6, 50.0)</td>
</tr>
<tr>
<td valign="top" align="left">S16</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">76</td>
<td valign="top" align="center">98.7</td>
<td valign="top" align="center">2.2</td>
<td valign="top" align="center">(94.7, 100.0)</td>
<td valign="top" align="center">73.7</td>
<td valign="top" align="center">5.2</td>
<td valign="top" align="center">(63.2, 82.9)</td>
</tr>
<tr>
<td valign="top" align="left">S17</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">40</td>
<td valign="top" align="center">100.0</td>
<td valign="top" align="center">0.0</td>
<td valign="top" align="center">(100.0, 100.0)</td>
<td valign="top" align="center">55.0</td>
<td valign="top" align="center">8.3</td>
<td valign="top" align="center">(37.5, 70.0)</td>
</tr>
<tr>
<td valign="top" align="left">S19</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">52</td>
<td valign="top" align="center">94.2</td>
<td valign="top" align="center">3.7</td>
<td valign="top" align="center">(86.5, 98.1)</td>
<td valign="top" align="center">71.2</td>
<td valign="top" align="center">6.7</td>
<td valign="top" align="center">(57.7, 82.7)</td>
</tr>
<tr>
<td valign="top" align="left">S21</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">34</td>
<td valign="top" align="center">97.1</td>
<td valign="top" align="center">5.2</td>
<td valign="top" align="center">(88.2, 100.0)</td>
<td valign="top" align="center">48.5</td>
<td valign="top" align="center">8.2</td>
<td valign="top" align="center">(29.4, 67.6)</td>
</tr>
<tr>
<td valign="top" align="left">S22</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">63</td>
<td valign="top" align="center">77.8</td>
<td valign="top" align="center">6.6</td>
<td valign="top" align="center">(66.7, 87.3)</td>
<td valign="top" align="center">87.3</td>
<td valign="top" align="center">4.6</td>
<td valign="top" align="center">(77.8, 95.2)</td>
</tr>
<tr>
<td valign="top" align="left">S24</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">57</td>
<td valign="top" align="center">87.7</td>
<td valign="top" align="center">4.6</td>
<td valign="top" align="center">(78.9, 94.7)</td>
<td valign="top" align="center">12.3</td>
<td valign="top" align="center">4.4</td>
<td valign="top" align="center">(5.3, 21.1)</td>
</tr>
<tr>
<td valign="top" align="left">S25</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">27</td>
<td valign="top" align="center">90.7</td>
<td valign="top" align="center">7.2</td>
<td valign="top" align="center">(77.8, 100.0)</td>
<td valign="top" align="center">66.7</td>
<td valign="top" align="center">8.1</td>
<td valign="top" align="center">(48.1, 81.5)</td>
</tr>
<tr>
<td valign="top" align="left">S28</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">44</td>
<td valign="top" align="center">85.2</td>
<td valign="top" align="center">5.1</td>
<td valign="top" align="center">(75.0, 93.2)</td>
<td valign="top" align="center">85.2</td>
<td valign="top" align="center">5.1</td>
<td valign="top" align="center">(75.0, 93.2)</td>
</tr>
<tr>
<td valign="top" align="left" colspan="4"><bold>Population mean</bold> <bold>&#x000B1;SD</bold></td>
<td valign="top" align="center"><bold>84.2</bold> <bold>&#x000B1;24.1</bold></td>
<td valign="top" align="center">&#x02014;</td>
<td valign="top" align="center">&#x02014;</td>
<td valign="top" align="center"><bold>66.7</bold> <bold>&#x000B1;29.7</bold></td>
<td valign="top" align="center">&#x02014;</td>
<td valign="top" align="center">&#x02014;</td>
</tr>
<tr>
<td valign="top" align="left" colspan="4"><bold>Range</bold></td>
<td valign="top" align="center"><bold>0.0&#x02013;100.0</bold></td>
<td valign="top" align="center">&#x02014;</td>
<td valign="top" align="center">&#x02014;</td>
<td valign="top" align="center"><bold>6.4&#x02013;100.0</bold></td>
<td valign="top" align="center">&#x02014;</td>
<td valign="top" align="center">&#x02014;</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Accuracy, SD, and CI computed via bootstrap resampling (10,000 iterations) using XGBoost with SET2 features (squeal-reduced dataset, SelectKBest). SD quantifies within-individual acoustic consistency.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Classification performance by demographic class with subject-level variability metrics.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Task</bold></th>
<th valign="top" align="center"><bold>Class</bold></th>
<th valign="top" align="center"><bold>Mean Acc (%)</bold></th>
<th valign="top" align="center"><bold>SD (%)</bold></th>
<th valign="top" align="center"><bold>95% CI lower</bold></th>
<th valign="top" align="center"><bold>95% CI upper</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Sex</td>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">85.8</td>
<td valign="top" align="center">7.7</td>
<td valign="top" align="center">68.7</td>
<td valign="top" align="center">96.4</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">79.5</td>
<td valign="top" align="center">2.4</td>
<td valign="top" align="center">75.1</td>
<td valign="top" align="center">83.6</td>
</tr>
<tr>
<td valign="top" align="left">Age</td>
<td valign="top" align="center">Adult</td>
<td valign="top" align="center">73.0</td>
<td valign="top" align="center">6.1</td>
<td valign="top" align="center">60.7</td>
<td valign="top" align="center">84.7</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">Juvenile</td>
<td valign="top" align="center">47.6</td>
<td valign="top" align="center">19.2</td>
<td valign="top" align="center">9.3</td>
<td valign="top" align="center">86.3</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Mean accuracy, SD, and 95% CI computed via bootstrap resampling over individuals (10,000 iterations). XGBoost with SET2 features (squeal-reduced dataset, SelectKBest).</p>
</table-wrap-foot>
</table-wrap>
<p>Sex classification achieved operationally viable accuracies ranging from 82% to 87% with Random Forest and XGBoost in the complete dataset, improving to 86%&#x02013;87% in the squeal-reduced variant (<xref ref-type="table" rid="T2">Table 2</xref>). Support Vector Machines matched or slightly exceeded ensemble methods (85%&#x02013;86%), demonstrating robust performance across feature representations. Random Forest and XGBoost exhibited comparable performance with moderate fold-level variability (SD = 5.8%&#x02013;7.7%), indicating consistent generalization across unseen individuals. Linear Discriminant Analysis dramatically underperformed (67%&#x02013;79%), though performance improved substantially with squeal removal (&#x0002B;8&#x02013;11 percentage points for SET2), suggesting violations of distributional assumptions exacerbated by acoustically heterogeneous call types.</p>
<p>Removing squeal-dominated subjects improved sex classification accuracies by 3&#x02013;6 percentage points for ensemble and SVM methods, with LDA showing the most dramatic gain (&#x0002B;11% for SET2), confirming that noisy broadband vocalizations disproportionately disrupt parametric classifiers. Per-class metrics (<xref ref-type="table" rid="T2">Table 2</xref>) revealed balanced performance across both sexes, with female recall (89%&#x02013;91%) slightly exceeding male recall (71%&#x02013;77%), though male precision remained competitive (75%&#x02013;82%). Subject-level bootstrap analysis revealed moderate between-individual variability (95% CI width: 12%&#x02013;18% across demographic classes; <xref ref-type="table" rid="T5">Table 5</xref>), indicating that while most individuals are consistently classified, a subset exhibits acoustically ambiguous signatures potentially reflecting behavioral flexibility or intermediate reproductive states.</p>
<p>Age classification proved substantially more challenging, with XGBoost achieving the highest accuracies (83%&#x02013;86%) but exhibiting severe juvenile under-detection (recall: 23%&#x02013;26%) despite high adult classification rates (recall: 97%&#x02013;98%; <xref ref-type="table" rid="T3">Table 3</xref>). Random Forest and SVM showed similar patterns but with even lower juvenile recall (11%&#x02013;22%), demonstrating that all methods systematically favor the majority adult class. Unlike sex classification, age accuracies showed dramatic benefit from squeal removal in XGBoost (&#x0002B;2&#x02013;3 percentage points) but minimal improvement for Random Forest (&#x0002B;3 percentage points), indicating that continuous developmental variation interacts complexly with call-type composition.</p>
<p>The addition of temporal-frequency and harmonic parameters (SET2: 38 features vs. SET1: 30 features) yielded modest and inconsistent effects. For sex classification, accuracies changed by &#x000B1;1&#x02013;2 percentage points across classifiers, with LDA showing the largest improvement (&#x0002B;8&#x02013;11 percentage points), likely reflecting increased feature space dimensionality stabilizing covariance matrix estimation. For age classification, SET2 produced marginal improvements (&#x0002B;1&#x02013;3 percentage points), with the largest gains in juvenile recall (&#x0002B;2&#x02013;4 percentage points for XGBoost and Random Forest), suggesting that explicit pitch parameters (<italic>f</italic><sub>0</sub>, frequency modulation, harmonics) provide limited additional discriminative power beyond formant-encoded MFCCs but slightly reduce adult-class bias through better juvenile characterization.</p>
<p>Recursive Feature Elimination (RFE) as an alternative to SelectKBest produced mixed results: sex classification accuracies decreased by 2&#x02013;6 percentage points with substantially reduced feature subsets (8&#x02013;15 features vs. 20&#x02013;28), while age classification improved marginally (&#x0002B;1&#x02013;3 percentage points for XGBoost), suggesting that age discrimination benefits from aggressive elimination of noisy or confounding predictors. However, the computational expense of RFE (3 &#x000D7; longer training time) limits its practicality for real-time deployment in autonomous monitoring systems.</p>
<p>Feature importance analysis provides mechanistic insight into the classification results. <xref ref-type="table" rid="T6">Tables 6</xref>, <xref ref-type="table" rid="T7">7</xref> list the top-ranked acoustic features for sex and age prediction, as determined by mean importance values in XGBoost across LOGO folds. For sex discrimination, low-order MFCCs and spectral skewness dominate, reflecting asymmetries in vocal tract shape and envelope. Age prediction relies more on mean <italic>f</italic><sub>0</sub>, number of harmonics, and certain chroma features, but with broadly distributed importance values&#x02014;consistent with the diffuse age separation observed in dimensionality reduction plots and limited age classification accuracy. These findings confirm that the main spectral patterns encoded by MFCCs and chromaticity provide robust sex information, while temporal-pitch features are insufficient for reliable age identification due to high within-class variability.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Top-ranked acoustic features for sex classification (XGBoost, SET2, squeal-reduced).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Feature</bold></th>
<th valign="top" align="center"><bold>Importance</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">MFCC1</td>
<td valign="top" align="center">0.18</td>
</tr>
<tr>
<td valign="top" align="left">MFCC2</td>
<td valign="top" align="center">0.14</td>
</tr>
<tr>
<td valign="top" align="left">Spectral skewness</td>
<td valign="top" align="center">0.12</td>
</tr>
<tr>
<td valign="top" align="left">Chroma 3</td>
<td valign="top" align="center">0.09</td>
</tr>
<tr>
<td valign="top" align="left">Chroma 9</td>
<td valign="top" align="center">0.08</td>
</tr>
<tr>
<td valign="top" align="left">Frequency modulation</td>
<td valign="top" align="center">0.07</td>
</tr>
<tr>
<td valign="top" align="left"><italic>F</italic><sub>0</sub> mean</td>
<td valign="top" align="center">0.05</td>
</tr>
<tr>
<td valign="top" align="left">Harmonics count</td>
<td valign="top" align="center">0.03</td>
</tr>
<tr>
<td valign="top" align="left">Duration</td>
<td valign="top" align="center">0.01</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Feature importance values averaged over LOGO folds and normalized to sum to 1.0.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Top-ranked acoustic features for age classification (XGBoost, SET2, squeal-reduced).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Feature</bold></th>
<th valign="top" align="center"><bold>Importance</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>F</italic><sub>0</sub> mean</td>
<td valign="top" align="center">0.22</td>
</tr>
<tr>
<td valign="top" align="left">Harmonics count</td>
<td valign="top" align="center">0.16</td>
</tr>
<tr>
<td valign="top" align="left">Chroma 3</td>
<td valign="top" align="center">0.12</td>
</tr>
<tr>
<td valign="top" align="left">Frequency modulation</td>
<td valign="top" align="center">0.09</td>
</tr>
<tr>
<td valign="top" align="left">MFCC2</td>
<td valign="top" align="center">0.08</td>
</tr>
<tr>
<td valign="top" align="left">Duration</td>
<td valign="top" align="center">0.07</td>
</tr>
<tr>
<td valign="top" align="left">MFCC1</td>
<td valign="top" align="center">0.06</td>
</tr>
<tr>
<td valign="top" align="left">Chroma 9</td>
<td valign="top" align="center">0.03</td>
</tr>
<tr>
<td valign="top" align="left">Spectral skewness</td>
<td valign="top" align="center">0.02</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Feature importance values averaged over LOGO folds and normalized to sum to 1.0.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<label>3.2</label>
<title>Subject-level performance and bootstrap confidence intervals</title>
<p>To quantify within-subject acoustic consistency and sources of classification uncertainty, we computed individual-level bootstrap confidence intervals (10,000 iterations) for all 16 subjects. Results are presented for the XGBoost classifier with SET2 features on the squeal-reduced dataset using SelectKBest feature selection&#x02014;the configuration that achieved optimal accuracy and minimized between-individual variability across all demographic tasks. This choice ensures that the reported variability primarily reflects biological heterogeneity rather than methodological artifacts, as this pipeline exhibited greater stability and lower standard deviations compared to alternative classifiers (Random Forest, SVM) or feature sets.</p>
<p>Subject-level analysis revealed substantial heterogeneity in classification consistency. For sex classification, three female adults (S03, S04, S17) achieved perfect accuracy (100%, SD = 0.0), indicating completely stereotyped sex-specific vocalizations. In stark contrast, S14 (female adult) exhibited 0% accuracy (SD = 0.0), systematically producing male-like calls across all 35 vocalizations. Intermediate performers showed variable uncertainty: S15 (male adult, 72.7%, SD = 11.0) exhibited the widest CI (50.0%&#x02013;90.9%), indicating high within-individual acoustic variability.</p>
<p>Age classification revealed even more dramatic individual-level patterns. Juvenile males showed bimodal consistency: S13 achieved only 6.4% accuracy (CI: 0.0%&#x02013;15.4%), indicating systematic production of adult-like vocalizations, while S22 and S28 reached 85%&#x02013;87% (CI: 75%&#x02013;95%), demonstrating consistent juvenile acoustic signatures. Female juvenile S24 also failed (12.3%, CI: 5.3%&#x02013;21.1%), contrasting sharply with near-perfect classification for adult females S03 (100%), S04 (96.8%), S10 (98.4%), and S14 (98.6%). Notably, S14&#x00027;s contrasting performance (0% for sex vs. 98.6% for age) indicates that her vocalizations encode age-typical acoustic features while deviating from female-typical spectral patterns.</p>
<p>Bootstrap-derived confidence intervals by demographic class (<xref ref-type="table" rid="T5">Table 5</xref>) quantified population-level uncertainty. Female classification (85.8%, 95% CI: 68.7%&#x02013;96.4%, width = 27.7%) exhibited substantially wider confidence bounds than males (79.5%, CI: 75.1%&#x02013;83.6%, width = 8.5%), reflecting greater between-individual acoustic variability among females. For age classification, juvenile CI spanned 9.3%&#x02013;86.3% (width = 77.0%)&#x02014;3.2 &#x000D7; wider than adults (60.7%&#x02013;84.7%, width = 24.0%)&#x02014;encompassing near-chance to near-perfect performance and confirming that current models cannot reliably generalize across unseen juvenile individuals.</p>
<p>Comparison with SET1 features showed that SET2&#x00027;s temporal-frequency augmentation systematically reduced within-individual SD by 10%&#x02013;25%, with population-level SD decreasing from 25.1 to 24.1% for sex and 31.2 to 29.7% for age classification, while mean accuracy remained statistically equivalent.</p>
</sec>
<sec>
<label>3.3</label>
<title>Acoustic feature space structure and demographic separability</title>
<p>Dimensionality reduction visualizations reveal contrasting demographic separability patterns. Sex classification exhibits moderate cluster separation in both PCA and t-SNE embeddings, with distinguishable male (blue) and female (red) distributions consistent with 85%&#x02013;87% accuracies (<xref ref-type="fig" rid="F2">Figure 2</xref>). Feature selection tightens cluster boundaries, confirming that univariate filtering retains discriminative information.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Sex categories exhibit moderate cluster separation in acoustic feature space. PCA (left) and t-SNE (right) embeddings from sex classification task reveal distinguishable male (blue) and female (red) clusters with partial overlap. <bold>(Top row)</bold> Full feature set (38 features); <bold>(bottom row)</bold> selected features (top 15 by univariate <italic>F</italic>-test). Points colored by four demographic classes show sex-dominated structure with weak age differentiation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1660388-g0002.tif">
<alt-text content-type="machine-generated">Four scatter plots comparing PCA and t-SNE methods for male and female data points. Top left: PCA with full features; bottom left: PCA with selected features. Top right: t-SNE with full features; bottom right: t-SNE with selected features. Blue dots represent male, red dots represent female.</alt-text>
</graphic>
</fig>
<p>In contrast, age categories show near-complete overlap across both methods (<xref ref-type="fig" rid="F3">Figure 3</xref>). PCA captures only 34%&#x02013;51% variance in first two components, and t-SNE manifolds fail to resolve age structure, directly explaining juvenile detection failure (26% recall with full features, 41% with selected).</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Age categories show substantial overlap in low-dimensional embeddings. PCA <bold>(left column)</bold> and t-SNE <bold>(right column)</bold> projections from the age classification task reveal broadly intermingled adult (green) and juvenile (orange) distributions, with only localized regions of partial separation. <bold>Top row</bold>: embeddings obtained from the full feature set; <bold>bottom row</bold>: embeddings obtained from the selected features. The similar large-scale structure across feature sets underscores the intrinsic difficulty of discriminating age classes from acoustic features alone.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1660388-g0003.tif">
<alt-text content-type="machine-generated">Four scatter plots comparing PCA and t-SNE analyses on age categories. Top left shows PCA with full features, illustrating a mix of green (adult) and orange (juvenile) points. Top right depicts t-SNE with full features showing more distinct clusters. Bottom left displays PCA using selected features, with overlapping green and orange points. Bottom right presents t-SNE with selected features, showing clearer clustering compared to PCA.</alt-text>
</graphic>
</fig>
<p>Joint 4-class visualizations expose sex-dominated hierarchical structure. When sex classification embeddings are colored by four demographic classes (<xref ref-type="fig" rid="F4">Figure 4</xref>), PCA and t-SNE show primary male-female separation with weak age substructure only within female clusters. Male juveniles and adults occupy identical acoustic regions across both feature sets, explaining systematic juvenile under-detection in males (0%&#x02013;14% recall).</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Sex-dominated hierarchical structure in 4-class demographic space. PCA <bold>(left)</bold> and t-SNE <bold>(right)</bold> embeddings calculated from sex classification task, colored by four demographic classes (male adult: dark blue; male juvenile: light blue; female adult: dark red; female juvenile: orange). <bold>(Top row)</bold> Full features; <bold>(bottom row)</bold> selected features. Primary male-female separation dominates across both feature sets. Weak age substructure visible only in female clusters; male age classes overlap completely.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1660388-g0004.tif">
<alt-text content-type="machine-generated">Four scatter plots compare data using PCA and t-SNE methods. Top-left: PCA with full features. Top-right: t-SNE with full features. Bottom-left: PCA with selected features. Bottom-right: t-SNE with selected features. Data points are color-coded by class: male adult, male juvenile, female adult, and female juvenile.</alt-text>
</graphic>
</fig>
<p>Conversely, age classification embeddings colored by four demographic classes (<xref ref-type="fig" rid="F5">Figure 5</xref>) confirm systematic overlap: male age classes form a single undifferentiated cluster across full and selected feature spaces, while female juveniles distribute across regions occupied by both female adults and male groups. Feature selection preserves this pattern, indicating that age-discriminative information is fundamentally limited rather than obscured by irrelevant features.</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Weak age separability across sex categories and feature sets. PCA <bold>(left column)</bold> and t-SNE <bold>(right column)</bold> embeddings from the age classification task are colored by four demographic classes: male adult (dark blue), male juvenile (light blue), female adult (dark red), and female juvenile (orange). <bold>Top row</bold>: embeddings obtained from the full feature set; <bold>bottom row</bold>: embeddings obtained from the selected features. Extensive overlap between male adult and male juvenile classes persists across feature sets and dimensionality reduction methods, whereas only partial separation is observed between female adults and juveniles, consistent with the observed classification asymmetries.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1660388-g0005.tif">
<alt-text content-type="machine-generated">Four scatter plots show data classification based on age and gender categories: Male Adult, Male Juvenile, Female Adult, Female Juvenile. Top plots use PCA and t-SNE with full features. Bottom plots use PCA and t-SNE with selected features. Points are color-coded by category and show varying clustering and spread across dimensions.</alt-text>
</graphic>
</fig>
<p>Three-dimensional t-SNE embedding with body size as the vertical axis (<xref ref-type="fig" rid="F6">Figure 6</xref>) reveals that juveniles occupy intermediate size zones (1.8&#x02013;2.4 m) overlapping with small adults, creating ambiguous acoustic-size combinations that confound age inference. Despite its high feature importance ranking, fundamental frequency shows negligible correlation with body size (<italic>R</italic><sup>2</sup> &#x0003D; 0.019, <italic>p</italic> &#x0003D; 0.56; <xref ref-type="fig" rid="F7">Figure 7</xref>). Juvenile <italic>f</italic><sub>0</sub> (2,520&#x02013;3,660 Hz) overlaps extensively with the adult range (1,000&#x02013;4,100 Hz), directly explaining systematic age misclassification. Notably, several subjects (S17, S18, S25, S27) presented median <italic>f</italic><sub>0</sub> estimates below 2,300 Hz, yet visual inspection of their spectrograms revealed energy concentration consistently above 2,300 Hz. Subject S14 (female adult, <italic>f</italic><sub>0</sub> = 4,040 Hz) exemplifies the dissociation between sex and age acoustic cues: her vocalizations achieve 0% sex accuracy but 98.6% age accuracy, indicating that elevated <italic>f</italic><sub>0</sub> encodes adult status independently of sex-typical spectral features.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Three-dimensional acoustic embedding stratified by body size. t-SNE projection with vertical axis representing body size (meters) shows size-stratified dispersion across four demographic classes (female adult, male adult, female juvenile, male juvenile). Substantial overlap in acoustic space, particularly between juvenile and adult males, explains low age classification accuracy despite successful sex discrimination.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1660388-g0006.tif">
<alt-text content-type="machine-generated">Three-dimensional t-SNE plot visualizing acoustic features and body size across four demographic classes: male adult, male juvenile, female adult, and female juvenile. Each class is represented by differently colored clusters, with axes labeled as t-SNE Dimension 1, 2, and 3.</alt-text>
</graphic>
</fig>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Fundamental frequency vs. body size reveals negligible developmental correlation. Median <italic>f</italic><sub>0</sub> plotted against body size for all subjects shows no significant relationship (<italic>R</italic><sup>2</sup> &#x0003D; 0.019, <italic>p</italic> &#x0003D; 0.56). Juvenile <italic>f</italic><sub>0</sub> range (2,520&#x02013;3,660 Hz) overlaps extensively with the adult range (1,000&#x02013;4,100 Hz), indicating that developmental vocal changes are either subtle or confounded by sex-specific and individual variation. Several subjects (S17, S18, S25, S27) display median <italic>f</italic><sub>0</sub> estimates below 2,300 Hz, yet visual spectrogram inspection reveals energy concentration above this threshold, suggesting systematic pitch estimation errors in certain vocalization types. This acoustic ambiguity explains systematic juvenile under-detection (recall: 14%&#x02013;26%) despite <italic>f</italic><sub>0</sub> ranking as the top age-discriminative feature in XGBoost models.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1660388-g0007.tif">
<alt-text content-type="machine-generated">Scatter plot titled &#x0201C;Median F&#x000E2;,&#x020AC; vs Body Size by Individual (4 demographic classes)&#x0201D;. It shows body size on the x-axis and median fundamental frequency on the y-axis. Symbols represent four demographics: male adult, male juvenile, female adult, and female juvenile. A dotted line shows a linear trend with R-squared value of 0.019, indicating weak correlation.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<label>3.4</label>
<title>Threshold optimization for juvenile detection</title>
<p>Standard decision thresholds (0.5 posterior probability) optimize global accuracy but systematically under-detect minority classes. We evaluated recall-precision trade-offs for age classification by varying thresholds from 0.1 to 0.9 (<xref ref-type="table" rid="T8">Table 8</xref>).</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Threshold optimization trade-offs for age classification (random forest, squeal-reduced SET1).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Threshold</bold></th>
<th valign="top" align="center"><bold>Juvenile recall (%)</bold></th>
<th valign="top" align="center"><bold>Adult recall (%)</bold></th>
<th valign="top" align="center"><bold>FPR (%)</bold></th>
<th valign="top" align="center"><bold>Global Acc (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">0.5 (default)</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">88</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">68.7</td>
</tr>
<tr>
<td valign="top" align="left">0.4</td>
<td valign="top" align="center">41</td>
<td valign="top" align="center">85</td>
<td valign="top" align="center">18</td>
<td valign="top" align="center">72.3</td>
</tr>
<tr>
<td valign="top" align="left">0.3</td>
<td valign="top" align="center">66</td>
<td valign="top" align="center">82</td>
<td valign="top" align="center">29</td>
<td valign="top" align="center">76.8</td>
</tr></tbody>
</table>
</table-wrap>
<p>Reducing the threshold to 0.3 increased juvenile recall from 14 to 66% (52 percentage point improvement), with false positives rising from 12 to 29%. An intermediate threshold of 0.4 provided more conservative balance (juvenile recall: 41%, adult recall: 85%, FPR: 18%).</p>
</sec>
<sec>
<label>3.5</label>
<title>Body size estimation from acoustic features</title>
<p>To evaluate continuous body size estimation as an alternative to categorical age classification, we trained Random Forest and XGBoost regressors on acoustic features alone under LOGO cross-validation (<xref ref-type="table" rid="T9">Table 9</xref>). The best-performing configuration was an ensemble model combining Random Forest and XGBoost predictions on SET1 spectral-cepstral features, which achieved MAE = 0.208 m, RMSE = 0.279 m, and <italic>R</italic><sup>2</sup> &#x0003D; 0.33, explaining 33% of body length variance across the observed range (1.70&#x02013;3.00 m).</p>
<table-wrap position="float" id="T9">
<label>Table 9</label>
<caption><p>Body size estimation performance via acoustic-only regression (LOGO cross-validation).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Feature set</bold></th>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>MAE (m)</bold></th>
<th valign="top" align="center"><bold>RMSE (m)</bold></th>
<th valign="top" align="center"><bold><italic>R</italic><sup>2</sup></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SET1 (30 features)</td>
<td valign="top" align="left">Random Forest</td>
<td valign="top" align="center">0.209</td>
<td valign="top" align="center">0.279</td>
<td valign="top" align="center">0.33</td>
</tr>
<tr>
<td valign="top" align="left">SET1 (30 features)</td>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="center">0.215</td>
<td valign="top" align="center">0.288</td>
<td valign="top" align="center">0.29</td>
</tr>
<tr>
<td valign="top" align="left">SET1 (30 features)</td>
<td valign="top" align="left">Ensemble</td>
<td valign="top" align="center">0.208</td>
<td valign="top" align="center">0.279</td>
<td valign="top" align="center">0.33</td>
</tr>
<tr>
<td valign="top" align="left">SET2 (38 features)</td>
<td valign="top" align="left">Random Forest</td>
<td valign="top" align="center">0.213</td>
<td valign="top" align="center">0.287</td>
<td valign="top" align="center">0.28</td>
</tr>
<tr>
<td valign="top" align="left">SET2 (38 features)</td>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="center">0.219</td>
<td valign="top" align="center">0.293</td>
<td valign="top" align="center">0.26</td>
</tr>
<tr>
<td valign="top" align="left">SET2 (38 features)</td>
<td valign="top" align="left">Ensemble</td>
<td valign="top" align="center">0.212</td>
<td valign="top" align="center">0.287</td>
<td valign="top" align="center">0.28</td>
</tr></tbody>
</table>
</table-wrap>
<p>Temporal-frequency augmentation (SET2) provided negligible improvement over spectral-cepstral features alone (SET1), with MAE differences of &#x02264; 0.004 m. Subject-level aggregation revealed heterogeneous performance: some individuals achieved MAE &#x0003C; 0.15 m, while others exceeded 0.40 m (<xref ref-type="table" rid="T4">Table 4</xref>).</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<label>4</label>
<title>Discussion</title>
<p>Our results demonstrate that sex classification from Greater Caribbean manatee vocalizations achieves operationally viable performance (85%&#x02013;87% accuracy, 75%&#x02013;78% macro-F1), while age classification presents fundamental challenges. Subject-level bootstrap analysis reveals substantial individual heterogeneity underlying population-level metrics, with critical implications for operational deployment.</p>
<sec>
<label>4.1</label>
<title>Acoustic basis of sex classification and individual variability</title>
<p>Sex classification reliability stems from spectral envelope characteristics (MFCCs, spectral skewness) rather than absolute pitch. Feature importance analysis confirms low-order MFCCs dominate discrimination, while temporal-frequency parameters provide modest supplementary contributions. However, bootstrap-derived 95% confidence intervals reveal female classification spans 68.7%&#x02013;96.4% (CI width: 27.7%), with extreme individual cases: three females achieved perfect classification while subject S14 failed completely (0% accuracy, systematically classified as male). Males showed narrower population CI (75.1%&#x02013;83.6%, width: 8.5%) but variable within-subject consistency [e.g., S15: CI (50.0%&#x02013;90.9%)].</p>
<p>This 3.3 &#x000D7; wider confidence interval for females compared to males indicates substantial between-individual acoustic heterogeneity within the female class. This variability may reflect underlying biological factors such as reproductive state fluctuations, maternal care behaviors, or inherent sex-specific differences in vocal plasticity&#x02014;influences documented in other marine mammals but unexplored in Greater Caribbean manatees. In contrast, male vocalizations exhibited greater stereotypy, with more stable acoustic signatures less influenced by behavioral or physiological variability, facilitating reliable classification in passive monitoring contexts.</p>
<p>These patterns indicate that while sex classification is operationally viable for population-level monitoring, approximately 10%&#x02013;15% of individuals will be systematically misclassified due to individual idiosyncrasies that override population-level patterns. Deployment protocols must archive probability scores rather than binary classifications to enable per-individual confidence assessment. Future work should investigate whether female acoustic heterogeneity correlates with observable demographic or reproductive variables, potentially enabling stratified classification models that account for intra-class diversity.</p>
</sec>
<sec>
<label>4.2</label>
<title>Fundamental challenges in age classification</title>
<p>Despite achieving 73%&#x02013;85% accuracy, age classification models exhibited severe juvenile under-detection (recall: 14%&#x02013;26%). Bootstrap 95% confidence intervals highlight this instability: the juvenile CI spans 9.3%&#x02013;86.3% (width: 77.0%), over three times wider than for adults (60.7%&#x02013;84.7%). This pronounced uncertainty results from multiple compounding factors.</p>
<p>First, a limited juvenile sample size (<italic>n</italic> &#x0003D; 4, only one female) means each LOGO fold tests on a single individual, amplifying subject-specific effects. Second, considerable class imbalance (adult:juvenile = 2.5:1) leads to a consistent bias toward the adult class, even after SMOTE oversampling. Third, median fundamental frequency (<italic>f</italic><sub>0</sub>)&#x02014;the top-ranked feature for age discrimination&#x02014;shows negligible correlation with body size (<italic>R</italic><sup>2</sup> &#x0003D; 0.019, <italic>p</italic> &#x0003D; 0.56), and juvenile <italic>f</italic><sub>0</sub> values (2,520&#x02013;3,660 Hz) overlap extensively with adults7 (1,000&#x02013;4,100 Hz; <xref ref-type="fig" rid="F7">Figure 7</xref>). This weak relationship is compounded by systematic pitch estimation errors: visual inspection revealed that subjects S17, S18, S25, and S27 display median <italic>f</italic><sub>0</sub> estimates below 2,300 Hz despite spectrogram energy concentration above this threshold, indicating that the PYIN estimator underestimates <italic>f</italic><sub>0</sub> in noisy or aperiodic vocalizations. Dimensionality reduction (<xref ref-type="fig" rid="F3">Figure 3</xref>) further confirms overlapping age distributions, in contrast to the moderate separation seen for sex (<xref ref-type="fig" rid="F2">Figure 2</xref>). Joint 4-class embeddings indicate sex dominates the acoustic feature space: male juveniles and adults cluster together, while female age classes show only weak separation.</p>
<p>Individual case studies reinforce this ambiguity. Subject S14 (female adult, <italic>f</italic><sub>0</sub> = 4,040 Hz) is classified with 0% sex accuracy but 98.6% age accuracy, indicating some independence of sex and age cues. In contrast, S13 (male juvenile, <italic>f</italic><sub>0</sub> = 3,660 Hz) is systematically misclassified as adult (6.4% age accuracy). Thus, elevated <italic>f</italic><sub>0</sub> alone is insufficient to reliably encode juvenile status, as both individual variation and sex-related baselines obscure developmental patterns.</p>
</sec>
<sec>
<label>4.3</label>
<title>Body size estimation and threshold optimization</title>
<sec>
<label>4.3.1</label>
<title>Continuous size estimation as alternative to categorical age</title>
<p>Given severe juvenile under-detection (recall: 14%&#x02013;26% at standard thresholds), acoustic body size regression offers an alternative that avoids imposing discrete age boundaries on continuous developmental variation. Moderate performance (MAE = 0.208 m, <italic>R</italic><sup>2</sup> &#x0003D; 0.33) is consistent with vocal tract scaling principles but leaves 67% of variance unexplained. The error (7% of typical body lengths) is comparable to visual estimation uncertainty but insufficient to reliably distinguish overlapping size classes (e.g., large juveniles vs. small adults).</p>
<p>Temporal-frequency features provided negligible improvement over spectral-cepstral features alone, suggesting explicit pitch parameters are either redundant with formant-encoded MFCCs or too behaviorally variable to enhance prediction&#x02014;consistent with weak <italic>f</italic><sub>0</sub>-body size correlation and pitch estimation artifacts discussed previously. Subject-level heterogeneity (MAE: 0.15&#x02013;0.40 m) mirrors patterns in age classification, indicating that vocal plasticity, repertoire composition, or individual acoustic idiosyncrasies constrain demographic inference independent of true class membership.</p>
<p>Despite limitations for precise individual measurements, this approach may support coarse size profiling into broad categories (small: &#x0003C; 2.3 m, medium: 2.3&#x02013;2.8 m, large: &#x0003E;2.8 m) when integrated with sex classification, providing operational value for demographic structure assessment in passive monitoring contexts without requiring categorical age assignment.</p></sec>
<sec>
<label>4.3.2</label>
<title>Threshold optimization for deployment</title>
<p>Threshold optimization reveals critical trade-offs for operational deployment. Lowering the threshold to 0.3 increases juvenile recall to 66% but raises false positives to 29%, suitable for surveillance prioritizing juvenile presence detection over precision (<xref ref-type="bibr" rid="B52">Stowell et al., 2019</xref>; <xref ref-type="bibr" rid="B25">Kahl et al., 2021</xref>). An intermediate threshold (0.4: 41% juvenile recall, 18% FPR) balances sensitivity and specificity for abundance monitoring (<xref ref-type="bibr" rid="B44">Rhinehart et al., 2020</xref>). Such optimization is critical in conservation contexts where missing endangered individuals (false negatives) incurs greater cost than false alarms (<xref ref-type="bibr" rid="B26">Kalan et al., 2015</xref>). Sex classification requires minimal threshold adjustment, achieving balanced performance at default settings.</p>
</sec>
</sec>
<sec>
<label>4.4</label>
<title>Study limitations and operational implications</title>
<p>Data span nearly three years (January 2021&#x02013;May 2023) across both dry and wet seasons in the Changuinola River (Bocas del Toro, Panama), providing temporal robustness. However, three limitations constrain generalization: (1) limited juvenile sample with extreme sex imbalance (one female, three males)&#x02014;a common constraint in endangered species monitoring where ethical considerations limit data collection intensity (<xref ref-type="bibr" rid="B57">Wrege et al., 2017</xref>; <xref ref-type="bibr" rid="B33">Measey et al., 2017</xref>), (2) binary age categorization imposing artificial boundaries on continuous ontogenetic development, and (3) task-specific optimal features&#x02014;RFE improved age (&#x0002B;2&#x02013;6 points) but degraded sex (&#x02013;2 to 10 points), indicating that age discrimination benefits from aggressive noise elimination while sex discrimination requires broader feature representation.</p>
<p>Despite limitations, sex-ratio monitoring is immediately deployable. Random Forest/XGBoost on spectral-cepstral features (SET1) provide robust classification (86%&#x02013;87%) without noise-sensitive pitch tracking&#x02014;critical for turbid riverine environments. For age classification, threshold tuning to operational objectives is essential, with explicit acknowledgment of 10%&#x02013;50% uncertainty depending on class. Deployment protocols should implement call-type routing, archive probability scores for uncertainty quantification, and report CI-based confidence bounds.</p>
<p>For Greater Caribbean manatees, acoustic monitoring addresses critical gaps in visual survey capacity in turbid tropical rivers (<xref ref-type="bibr" rid="B46">Sanchez-Galan et al., 2025</xref>; <xref ref-type="bibr" rid="B20">Guzman et al., 2025</xref>). Our bootstrap-derived confidence intervals provide the statistical foundation for uncertainty-aware conservation decision-making essential when acoustic data inform management actions for endangered populations. Future work should prioritize balanced juvenile sampling, longitudinal tracking for developmental trajectories, deep learning approaches (CNNs on spectrograms) to automatically learn discriminative features, and integration with individual identification frameworks for acoustic mark-recapture&#x02014;enabling simultaneous estimation of abundance, sex ratios, and demographic structure from long-term hydrophone deployments.</p>
</sec>
<sec>
<label>4.5</label>
<title>Pitch estimation artifacts in heterogeneous vocalizations</title>
<p>Visual spectrogram inspection revealed systematic discrepancies for subjects S17, S18, S25, and S27: median <italic>f</italic><sub>0</sub> estimates fell below 2,300 Hz despite energy concentration above this threshold. This indicates that autocorrelation-based pitch estimators may be tracking residual low-frequency noise near the high-pass filter cutoff (1 kHz) rather than the true vocal fundamental, particularly in noisy, aperiodic, or broadband vocalizations where harmonic structure is ambiguous (<xref ref-type="bibr" rid="B31">Mauch and Dixon, 2014</xref>). This artifact explains the weak correlation between estimated <italic>f</italic><sub>0</sub> and body size (<italic>R</italic><sup>2</sup> &#x0003D; 0.019) despite biomechanical predictions <xref ref-type="bibr" rid="B15">Fitch, (1997</xref>).</p>
<p>This measurement error explains two key findings: (1) the 3&#x02013;8 percentage point improvement when squeal-dominated subjects were excluded, and (2) why <italic>f</italic><sub>0</sub> ranks as highly important yet fails as a reliable age predictor. Future work should implement stricter preprocessing (e.g., higher cutoff frequencies or adaptive filtering) and call-type classification as preprocessing (<xref ref-type="bibr" rid="B47">Schneider et al., 2024</xref>), extracting pitch parameters only from tonal calls with unambiguous harmonic structure while using alternative spectral descriptors (MFCCs, centroid, bandwidth) for noisy vocalizations.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusions</title>
<p>Acoustic sex classification from Greater Caribbean manatee vocalizations achieves operationally viable performance (85%&#x02013;87% accuracy) suitable for passive monitoring, though subject-level bootstrap analysis reveals substantial individual heterogeneity (female 95% CI: 68.7%&#x02013;96.4%; male: 75.1%&#x02013;83.6%), indicating 10%&#x02013;15% of individuals will be systematically misclassified. Operational deployment must incorporate individual-level confidence bounds rather than relying solely on population metrics.</p>
<p>Age classification presents fundamental challenges despite 73%&#x02013;85% global accuracy, with severe juvenile under-detection (14%&#x02013;26% recall) and extreme uncertainty (juvenile 95% CI: 9.3%&#x02013;86.3%, 3.2 &#x000D7; wider than adults). Fundamental frequency shows negligible correlation with body size (<italic>R</italic><sup>2</sup> &#x0003D; 0.019), with juvenile and adult <italic>f</italic><sub>0</sub> ranges overlapping extensively, preventing discrete boundaries. Threshold optimization improves juvenile detection to 63% but elevates false positives to 37%, representing context-dependent trade-offs for conservation surveillance. Body size estimation via acoustic regression achieved proof-of-concept (MAE = 0.208 m, <italic>R</italic><sup>2</sup> &#x0003D; 0.33) supporting coarse profiling into broad categories when integrated with sex classification.</p>
<p>Sex-ratio monitoring provides immediate conservation value for endangered populations in turbid riverine habitats. Data spanning three years across both seasons demonstrate temporal robustness. Age classification requires expanded sampling prioritizing demographic balance (currently <italic>n</italic> &#x0003D; 4 juveniles, 1 female) and longitudinal tracking to characterize vocal ontogeny. Integration of demographic classification with established individual identification frameworks would enable comprehensive acoustic mark-recapture, simultaneously estimating abundance, sex ratios, and demographic structure&#x02014;complementing ongoing monitoring efforts in the Changuinola River. This study establishes both feasibility and critical limitations of acoustic demographic inference, providing bootstrap-derived uncertainty metrics essential for evidence-based conservation decision-making.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The animal study was approved by Animal Care and Use Committee (ACUC), Smithsonian Tropical Research Institute (STRI). The study was conducted in accordance with the local legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>FM: Project administration, Methodology, Data curation, Conceptualization, Writing &#x02013; review &#x00026; editing. KC: Visualization, Methodology, Data curation, Software, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing, Conceptualization. HP: Writing &#x02013; review &#x00026; editing. RE: Data curation, Project administration, Writing &#x02013; review &#x00026; editing. HG: Data curation, Conceptualization, Project administration, Writing &#x02013; review &#x00026; editing, Methodology. JS-G: Writing &#x02013; review &#x00026; editing, Methodology, Conceptualization.</p>
</sec>
<ack><title>Acknowledgments</title><p>The authors thank the Government of Panama, through the Ministry of the Environment, for providing the research permits (SE/A-79-2019; ARG-004-2023). We thank Jossio Guillen and Carlos Guevara for unconditional field assistance and logistical support. We thank Cristal C&#x000E1;ceres for valuable discussions regarding the use of chroma features for the analysis of manatee vocalizations. We thank Alfredo Caballero and Roberto Gonzalez for their assistance in installing the pen and Candy Real for providing transportation support. We also thank the AAMVECONA board for granting us access to the pier and electricity to observe the manatees. The authors also thank the board of COOBANA R. L. Banana Company, particularly Chito Quintero, Diomedes Rodriguez, and Dinora Beitia, for providing bananas at no cost for over two years. The authors acknowledge administrative support provided by CEMCIT-AIP, Smithsonian Tropical Research (STRI), and Universidad Tecnol&#x000F3;gica de Panam&#x000E1; (UTP). We thank Maristela Nuques for her support and for granting access to her property during the manatee capture work.</p></ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was used in the creation of this manuscript. Generative AI tools were used to assist the author(s) in improving the English writing, as well as in programming, code organization, and results analysis.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Altaf</surname> <given-names>M.</given-names></name> <name><surname>Rahman</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). &#x0201C;Age classification based on voice using MEL-spectrogram and convolutional neural networks," in <italic>2023 International Conference on Digital Signal Processing (DSP)</italic> (Rhodes).</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arlot</surname> <given-names>S.</given-names></name> <name><surname>Celisse</surname> <given-names>A.</given-names></name></person-group> (<year>2010</year>). <article-title>A survey of cross-validation procedures for model selection</article-title>. <source>Stat. Surv</source>. <volume>4</volume>, <fpage>40</fpage>&#x02013;<lpage>79</lpage>. doi: <pub-id pub-id-type="doi">10.1214/09-SS054</pub-id></mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brady</surname> <given-names>A. G.</given-names></name> <name><surname>Alves</surname> <given-names>L. C. P. S.</given-names></name> <name><surname>Moron</surname> <given-names>J.</given-names></name> <name><surname>da Silva</surname> <given-names>V. M. F.</given-names></name> <name><surname>Mann</surname> <given-names>D. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Acoustic contour and body size in calves of the Amazonian and West Indian manatees</article-title>. <source>Sci. Rep</source>. <volume>12</volume>:<fpage>19574</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-022-23321-7</pub-id></mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brady</surname> <given-names>A. G.</given-names></name> <name><surname>Nowacek</surname> <given-names>D. P.</given-names></name> <name><surname>Mann</surname> <given-names>D. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Vocalizations of the West Indian manatee (<italic>Trichechus manatus latirostris</italic>): call types and acoustic features</article-title>. <source>J. Acoust. Soc. Am</source>. <volume>147</volume>, <fpage>1925</fpage>&#x02013;<lpage>1935</lpage>. doi: <pub-id pub-id-type="doi">10.1121/10.0000849</pub-id></mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brady</surname> <given-names>E. A.</given-names></name> <name><surname>Bonde</surname> <given-names>R. K.</given-names></name> <name><surname>Schopmeyer</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). Behavior related vocalizations of the <italic>Florida manatee (Trichechus manatus latirostris</italic>). <italic>J. Mammal</italic>. <volume>103</volume>, <fpage>389</fpage>&#x02013;<lpage>400</lpage>. doi: <pub-id pub-id-type="doi">10.1111/mms.12904</pub-id></mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name></person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn</source>. <volume>45</volume>, <fpage>5</fpage>&#x02013;<lpage>32</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Castelblanco-Mart&#x000ED;nez</surname> <given-names>D. N.</given-names></name> <name><surname>Nourisson</surname> <given-names>C.</given-names></name> <name><surname>Quintana-Rizzo</surname> <given-names>E.</given-names></name> <name><surname>Padilla-Saldivar</surname> <given-names>J.</given-names></name> <name><surname>Schmitter-Soto</surname> <given-names>J. J.</given-names></name></person-group> (<year>2012</year>). <article-title>Potential effects of human pressure and habitat fragmentation on population viability of the <italic>Antillean manatee Trichechus manatus manatus</italic>: a predictive model</article-title>. <source>Endanger. Species Res</source>. <volume>18</volume>, <fpage>129</fpage>&#x02013;<lpage>145</lpage>. doi: <pub-id pub-id-type="doi">10.3354/esr00439</pub-id></mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chawla</surname> <given-names>N. V.</given-names></name> <name><surname>Bowyer</surname> <given-names>K. W.</given-names></name> <name><surname>Hall</surname> <given-names>L. O.</given-names></name> <name><surname>Kegelmeyer</surname> <given-names>W. P.</given-names></name></person-group> (<year>2002</year>). <article-title>Smote: synthetic minority oversampling technique</article-title>. <source>J. Artif. Intell. Res</source>. <volume>16</volume>, <fpage>321</fpage>&#x02013;<lpage>357</lpage>. doi: <pub-id pub-id-type="doi">10.1613/jair.953</pub-id></mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Guestrin</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). &#x0201C;Xgboost: a scalable tree boosting system," in <italic>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</italic> (New York, NY: ACM), <fpage>785</fpage>&#x02013;<lpage>794</lpage>. doi: <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cortes</surname> <given-names>C.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>1995</year>). <article-title>Support-vector networks</article-title>. <source>Mach. Learn</source>. <volume>20</volume>, <fpage>273</fpage>&#x02013;<lpage>297</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1022627411411</pub-id></mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Diaz-Ferguson</surname> <given-names>E.</given-names></name> <name><surname>Guzman</surname> <given-names>M.</given-names></name> <name><surname>Hunter</surname> <given-names>H. M.</given-names></name></person-group> (<year>2017</year>). <article-title>Genetic composition and connectivity of the West Indian <italic>Antillean manatee (Trichechus manatus manatus</italic>) in Panama</article-title>. <source>Aquat. Mamm</source>. <volume>43</volume>, <fpage>378</fpage>&#x02013;<lpage>386</lpage>. doi: <pub-id pub-id-type="doi">10.1578/AM.43.4.2017.378</pub-id></mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Efron</surname> <given-names>B.</given-names></name> <name><surname>Tibshirani</surname> <given-names>R. J.</given-names></name></person-group> (<year>1994</year>). <source>An introduction to the bootstrap.</source> <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Chapman and Hall/CRC</publisher-name>. doi: <pub-id pub-id-type="doi">10.1007/978-1-4899-4541-9</pub-id></mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Erbe</surname> <given-names>C.</given-names></name> <name><surname>Reichmuth</surname> <given-names>C.</given-names></name> <name><surname>Cunningham</surname> <given-names>K.</given-names></name> <name><surname>Lucke</surname> <given-names>K.</given-names></name> <name><surname>Dooling</surname> <given-names>R.</given-names></name></person-group> (<year>2013</year>). <article-title>Communication masking in marine mammals: a review and research strategy</article-title>. <source>Mar. Pollut. Bull</source>. <volume>103</volume>, <fpage>15</fpage>&#x02013;<lpage>38</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.marpolbul.2015.12.007</pub-id><pub-id pub-id-type="pmid">26707982</pub-id></mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fisher</surname> <given-names>R. A.</given-names></name></person-group> (<year>1936</year>). <article-title>The use of multiple measurements in taxonomic problems</article-title>. <source>Ann. Eugen</source>. <volume>7</volume>, <fpage>179</fpage>&#x02013;<lpage>188</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1469-1809.1936.tb02137.x</pub-id></mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fitch</surname> <given-names>W. T.</given-names></name></person-group> (<year>1997</year>). <article-title>Vocal tract length and formant frequency dispersion correlate with body size in rhesus macaques</article-title>. <source>J. Acoust. Soc. Am</source>. <volume>102</volume>, <fpage>1213</fpage>&#x02013;<lpage>1222</lpage>. doi: <pub-id pub-id-type="doi">10.1121/1.421048</pub-id><pub-id pub-id-type="pmid">9265764</pub-id></mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Garc&#x000ED;a</surname> <given-names>N. C.</given-names></name> <name><surname>M&#x000F8;ller</surname> <given-names>A. P.</given-names></name> <name><surname>Tubaro</surname> <given-names>P. L.</given-names></name> <name><surname>Gordo</surname> <given-names>O.</given-names></name></person-group> (<year>2018</year>). <article-title>Dissecting the roles of body size and beak morphology in song evolution in tanagers (Thraupidae)</article-title>. <source>Auk: Ornithol. Adv</source>. <volume>135</volume>, <fpage>262</fpage>&#x02013;<lpage>275</lpage>. doi: <pub-id pub-id-type="doi">10.1642/AUK-17-146.1</pub-id></mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guyon</surname> <given-names>I.</given-names></name> <name><surname>Elisseeff</surname> <given-names>A.</given-names></name></person-group> (<year>2003</year>). <article-title>An introduction to variable and feature selection</article-title>. <source>J. Mach. Learn. Res</source>. <volume>3</volume>, <fpage>1157</fpage>&#x02013;<lpage>1182</lpage>. doi: <pub-id pub-id-type="doi">10.1162/153244303322753616</pub-id></mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guyon</surname> <given-names>I.</given-names></name> <name><surname>Weston</surname> <given-names>J.</given-names></name> <name><surname>Barnhill</surname> <given-names>S.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>2002</year>). <article-title>Gene selection for cancer classification using support vector machines</article-title>. <source>Mach. Learn</source>. <volume>46</volume>, <fpage>389</fpage>&#x02013;<lpage>422</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1012487302797</pub-id></mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guzman</surname> <given-names>H. M.</given-names></name> <name><surname>Condit</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Abundance of manatees in Panama estimated from side-scan sonar</article-title>. <source>Wildl. Soc. Bull</source>. <volume>41</volume>, <fpage>556</fpage>&#x02013;<lpage>565</lpage>. doi: <pub-id pub-id-type="doi">10.1002/wsb.793</pub-id></mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guzman</surname> <given-names>H. M.</given-names></name> <name><surname>Est&#x000E9;vez</surname> <given-names>R. M.</given-names></name> <name><surname>Contreras</surname> <given-names>K.</given-names></name> <name><surname>Poveda</surname> <given-names>H.</given-names></name> <name><surname>Merchan</surname> <given-names>F.</given-names></name> <name><surname>Sanchez-Galan</surname> <given-names>J. E.</given-names></name></person-group> (<year>2025</year>). <article-title>Year-round residency and movement behavior of greater Caribbean manatees (<italic>Trichechus manatus manatus</italic>) in Panama and Costa Rica</article-title>. <source>Front. Mar. Sci</source>. <volume>12</volume>:<fpage>1661294</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmars.2025.1661294</pub-id></mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Hartman</surname> <given-names>D. S.</given-names></name></person-group> (<year>1979</year>). <source>Ecology and Behavior of the Manatee (Trichechus manatus) in Florida</source>. <publisher-loc>Special Publication No. 5. Pittsburgh, PA</publisher-loc>: <publisher-name>American Society of Mammalogists</publisher-name>. doi: <pub-id pub-id-type="doi">10.5962/bhl.title.39474</pub-id></mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Hines</surname> <given-names>E. M.</given-names> <suffix>III.</suffix></name> <name><surname>Aragones</surname> <given-names>J. E. R. L. V.</given-names></name> <name><surname>Mignucci-Giannoni</surname> <given-names>A. A.</given-names></name> <name><surname>Marmontel</surname> <given-names>M. (Eds.)</given-names></name></person-group> (<year>2012</year>). <source>Sirenian Conservation: Issues and Strategies in Developing Countries</source>. <publisher-loc>Gainesville, FL</publisher-loc>: <publisher-name>University Press of Florida</publisher-name>. doi: <pub-id pub-id-type="doi">10.2307/j.ctvx079z0</pub-id></mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ivanenko</surname> <given-names>A.</given-names></name> <name><surname>Watkins</surname> <given-names>P.</given-names></name> <name><surname>van Gerven</surname> <given-names>M. A. J.</given-names></name> <name><surname>Hammerschmidt</surname> <given-names>K.</given-names></name> <name><surname>Englitz</surname> <given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>Classifying sex and strain from mouse ultrasonic vocalizations using deep learning</article-title>. <source>PLoS Comput. Biol</source>. <volume>16</volume>:<fpage>e1007918</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1007918</pub-id><pub-id pub-id-type="pmid">32569292</pub-id></mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jung</surname> <given-names>D. H.</given-names></name> <name><surname>Kim</surname> <given-names>N. Y.</given-names></name> <name><surname>Moon</surname> <given-names>S. H.</given-names></name> <name><surname>Jhin</surname> <given-names>C.</given-names></name> <name><surname>Kim</surname> <given-names>H. J.</given-names></name> <name><surname>Yang</surname> <given-names>J. S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Deep learning-based cattle vocal classification model with potential for age, gender, and physiological state assessment</article-title>. <source>Animals</source> <volume>11</volume>:<fpage>456</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ani11020357</pub-id></mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kahl</surname> <given-names>S.</given-names></name> <name><surname>Wood</surname> <given-names>C. M.</given-names></name> <name><surname>Eibl</surname> <given-names>M.</given-names></name> <name><surname>Klinck</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>BirdNET: a deep learning solution for avian diversity monitoring</article-title>. <source>Ecol. Inform</source>. <volume>61</volume>:<fpage>101236</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ecoinf.2021.101236</pub-id></mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kalan</surname> <given-names>A. K.</given-names></name> <name><surname>Piel</surname> <given-names>A. K.</given-names></name> <name><surname>Mundry</surname> <given-names>R.</given-names></name> <name><surname>Wittig</surname> <given-names>R. M.</given-names></name> <name><surname>Boesch</surname> <given-names>C.</given-names></name> <name><surname>K&#x000FC;hl</surname> <given-names>H. S.</given-names></name></person-group> (<year>2015</year>). <article-title>Towards the automated detection and occupancy estimation of primates using passive acoustic monitoring</article-title>. <source>Ecol. Indic</source>. <volume>54</volume>, <fpage>217</fpage>&#x02013;<lpage>226</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ecolind.2015.02.023</pub-id></mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Kuhn</surname> <given-names>M.</given-names></name> <name><surname>Johnson</surname> <given-names>K.</given-names></name></person-group> (<year>2013</year>). <source>Applied Predictive Modeling</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name>. doi: <pub-id pub-id-type="doi">10.1007/978-1-4614-6849-3</pub-id></mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lefebvre</surname> <given-names>L. W.</given-names></name> <name><surname>Marmontel</surname> <given-names>M.</given-names></name> <name><surname>Reid</surname> <given-names>J. P.</given-names></name> <name><surname>Rathbun</surname> <given-names>G. B.</given-names></name> <name><surname>Domning</surname> <given-names>D. P.</given-names></name></person-group> (<year>2001</year>). &#x0201C;Status and biogeography of the West Indian manatee," in <source>Biogeography of the West Indies: Patterns and Perspectives</source>, eds. C. A. Woods, and F. E. Sergile, 2nd Edn (Boca Raton, FL: CRC Press), <fpage>425</fpage>&#x02013;<lpage>474</lpage>. doi: <pub-id pub-id-type="doi">10.1201/9781420039481.ch22</pub-id></mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Logan</surname> <given-names>B.</given-names></name></person-group> (<year>2000</year>). &#x0201C;Mel frequency cepstral coefficients for music modeling," in <italic>International Symposium on Music Information Retrieval (ISMIR)</italic> (Plymouth, MA), <fpage>1</fpage>&#x02013;<lpage>11</lpage>.</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Marmontel</surname> <given-names>M.</given-names></name></person-group> (<year>1995</year>). &#x0201C;Age and reproduction in female Florida manatees," in <source>Population Biology of the Florida Manatee, number 1 in Information and Technology Report</source>, eds. T. J. O&#x00027;Shea, B. B. Ackerman, and H. F. Percival (Washington, DC: U.S. Department of the Interior, National Biological Service), <fpage>98</fpage>&#x02013;<lpage>119</lpage>.</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mauch</surname> <given-names>M.</given-names></name> <name><surname>Dixon</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). &#x0201C;pYIN: a fundamental frequency estimator using probabilistic threshold distributions," in <italic>IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</italic> (Florence: IEEE), <fpage>659</fpage>&#x02013;<lpage>663</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICASSP.2014.6853678</pub-id></mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McFee</surname> <given-names>B.</given-names></name> <name><surname>Raffel</surname> <given-names>C.</given-names></name> <name><surname>Liang</surname> <given-names>D.</given-names></name> <name><surname>Ellis</surname> <given-names>D.</given-names></name> <name><surname>McVicar</surname> <given-names>M.</given-names></name> <name><surname>Battenberg</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2015</year>). &#x0201C;Librosa: audio and music signal analysis in Python," in <italic>Proceedings of the 14th Python in Science Conference, Vol. 8</italic> (Austin, TX), <fpage>18</fpage>&#x02013;<lpage>25</lpage>. doi: <pub-id pub-id-type="doi">10.25080/Majora-7b98e3ed-003</pub-id></mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Measey</surname> <given-names>J.</given-names></name> <name><surname>Stevenson</surname> <given-names>B. C.</given-names></name> <name><surname>Scott</surname> <given-names>T.</given-names></name> <name><surname>Altwegg</surname> <given-names>R.</given-names></name> <name><surname>Bormpoudakis</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Counting chirps: acoustic monitoring of cryptic frogs</article-title>. <source>J. Appl. Ecol.</source> <volume>54</volume>, <fpage>894</fpage>&#x02013;<lpage>902</lpage>. doi: <pub-id pub-id-type="doi">10.1111/1365-2664.12810</pub-id></mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Merchan</surname> <given-names>F.</given-names></name> <name><surname>Contreras</surname> <given-names>K.</given-names></name> <name><surname>Poveda</surname> <given-names>H.</given-names></name> <name><surname>Guzman</surname> <given-names>H. M.</given-names></name> <name><surname>Sanchez-Galan</surname> <given-names>J. E.</given-names></name></person-group> (<year>2024</year>). <article-title>Unsupervised identification of greater Caribbean manatees using scattering wavelet transform and hierarchical density clustering from underwater bioacoustics recordings</article-title>. <source>Front. Mar. Sci</source>. <volume>11</volume>:<fpage>1416247</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmars.2024.1416247</pub-id></mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Merchan</surname> <given-names>F.</given-names></name> <name><surname>Echevers</surname> <given-names>G.</given-names></name> <name><surname>Poveda</surname> <given-names>H.</given-names></name> <name><surname>Sanchez-Galan</surname> <given-names>J. E.</given-names></name> <name><surname>Guzman</surname> <given-names>H. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Detection and identification of manatee individual vocalizations in Panamanian wetlands using spectrogram clustering</article-title>. <source>J. Acoust. Soc. Am</source>. <volume>146</volume>, <fpage>1745</fpage>&#x02013;<lpage>1757</lpage>. doi: <pub-id pub-id-type="doi">10.1121/1.5126504</pub-id><pub-id pub-id-type="pmid">31590493</pub-id></mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Merchan</surname> <given-names>F.</given-names></name> <name><surname>Guerra</surname> <given-names>A.</given-names></name> <name><surname>Poveda</surname> <given-names>H.</given-names></name> <name><surname>Guzm&#x000E1;n</surname> <given-names>H. M.</given-names></name> <name><surname>Sanchezand -Galan</surname> <given-names>J. E.</given-names></name></person-group> (<year>2020</year>). <article-title>Bioacoustic classification of Antillean manatee vocalization spectrograms using deep convolutional neural networks</article-title>. <source>Appl. Sci</source>. <volume>10</volume>:<fpage>3286</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app10093286</pub-id></mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Miksis-Olds</surname> <given-names>J. L.</given-names></name> <name><surname>Martin</surname> <given-names>B.</given-names></name> <name><surname>Tyack</surname> <given-names>P. L.</given-names></name></person-group> (<year>2018</year>). <article-title>Exploring the ocean through soundscapes</article-title>. <source>Acoust. Today</source> <volume>14</volume>, <fpage>26</fpage>&#x02013;<lpage>34</lpage>.</mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Morales-Vela</surname> <given-names>B.</given-names></name> <name><surname>Quintana-Rizzo</surname> <given-names>E.</given-names></name> <name><surname>Mignucci-Giannoni</surname> <given-names>A. A.</given-names></name></person-group> (<year>2024</year>). <article-title><italic>Trichechus manatus</italic> ssp</article-title>. <source>manatus. The IUCN Red List of Threatened Species 2024: e.T22105A43793924</source>. <publisher-loc>Gland</publisher-loc>: <publisher-name>IUCN</publisher-name>.</mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>M&#x000FC;ller</surname> <given-names>M.</given-names></name> <name><surname>Ewert</surname> <given-names>S.</given-names></name></person-group> (<year>2007</year>). &#x0201C;Chroma toolbox: MATLAB implementations for extracting variants of chroma-based audio features," in <italic>Proceedings of the International Conference on Music Information Retrieval (ISMIR)</italic> (Vienna), <fpage>215</fpage>&#x02013;<lpage>216</lpage>.</mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>O&#x00027;Shea</surname> <given-names>T. J.</given-names></name> <name><surname>Poch&#x000E9;</surname> <given-names>L. B.</given-names></name></person-group> (<year>2006</year>). Aspects of underwater sound communication in Florida manatees (<italic>Trichechus manatus latirostris</italic>). <italic>J. Mammal</italic>. <volume>87</volume>, <fpage>1061</fpage>&#x02013;<lpage>1071</lpage>. doi: <pub-id pub-id-type="doi">10.1644/06-MAMM-A-066R1.1</pub-id></mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pourhoseingholi</surname> <given-names>M. A.</given-names></name> <name><surname>Baghestani</surname> <given-names>A. R.</given-names></name> <name><surname>Vahedi</surname> <given-names>M.</given-names></name></person-group> (<year>2012</year>). <article-title>How to control confounding effects by statistical analysis</article-title>. <source>Gastroenterol. Hepatol. Bed Bench</source> <volume>5</volume>, <fpage>79</fpage>&#x02013;<lpage>83</lpage>. doi: <pub-id pub-id-type="doi">10.22037/ghfbb.v5i2.246</pub-id><pub-id pub-id-type="pmid">24834204</pub-id></mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Provost</surname> <given-names>F.</given-names></name> <name><surname>Fawcett</surname> <given-names>T.</given-names></name></person-group> (<year>2001</year>). <article-title>Robust classification for imprecise environments</article-title>. <source>Mach. Learn</source>. <volume>42</volume>, <fpage>203</fpage>&#x02013;<lpage>231</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1007601015854</pub-id></mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Reynolds</surname> <given-names>J. E.</given-names></name> <name><surname>Odell</surname> <given-names>D. K.</given-names></name></person-group> (<year>1992</year>). <source>Manatees and Dugongs</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Facts on File</publisher-name>.</mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rhinehart</surname> <given-names>T. A.</given-names></name> <name><surname>Chronister</surname> <given-names>L. M.</given-names></name> <name><surname>Devlin</surname> <given-names>T.</given-names></name> <name><surname>Kitzes</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Acoustic localization of terrestrial wildlife: current practices and future opportunities</article-title>. <source>Ecol. Evol</source>. <volume>10</volume>, <fpage>6794</fpage>&#x02013;<lpage>6818</lpage>. doi: <pub-id pub-id-type="doi">10.1002/ece3.6216</pub-id><pub-id pub-id-type="pmid">32724552</pub-id></mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rycyk</surname> <given-names>A.</given-names></name> <name><surname>Bolaji</surname> <given-names>D. A.</given-names></name> <name><surname>Factheu</surname> <given-names>C.</given-names></name> <name><surname>Kamla Takoukam</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>Using transfer learning with a convolutional neural network to detect African manatee (<italic>Trichechus senegalensis</italic>) vocalizations</article-title>. <source>JASA Express Lett</source>. <volume>2</volume>:<fpage>101201</fpage>. doi: <pub-id pub-id-type="doi">10.1121/10.0016543</pub-id><pub-id pub-id-type="pmid">36586963</pub-id></mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sanchez-Galan</surname> <given-names>J. E.</given-names></name> <name><surname>Contreras</surname> <given-names>K.</given-names></name> <name><surname>Denoce</surname> <given-names>A.</given-names></name> <name><surname>Poveda</surname> <given-names>H.</given-names></name> <name><surname>Merchan</surname> <given-names>F.</given-names></name> <name><surname>Guzm&#x000E1;n</surname> <given-names>H. M.</given-names></name></person-group> (<year>2025</year>). <article-title>Drone-based detection and classification of greater Caribbean manatees in the panama canal basin</article-title>. <source>Drones</source> <volume>9</volume>:<fpage>230</fpage>. doi: <pub-id pub-id-type="doi">10.3390/drones9040230</pub-id></mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname> <given-names>S.</given-names></name> <name><surname>Dierkes</surname> <given-names>P. W.</given-names></name> <name><surname>Fersen</surname> <given-names>L. V.</given-names></name></person-group> (<year>2024</year>). <article-title>Automated detection and classification of Antillean manatee vocalizations using CNNS and clustering approaches</article-title>. <source>Front. Conserv. Sci</source>. <volume>5</volume>:<fpage>1405243</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fcosc.2024.1405243</pub-id></mixed-citation>
</ref>
<ref id="B48">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Searle</surname> <given-names>S. R.</given-names></name> <name><surname>Speed</surname> <given-names>F. M.</given-names></name> <name><surname>Milliken</surname> <given-names>G. A.</given-names></name></person-group> (<year>2017</year>). <source>Population Marginal Means in the Linear Model: An Alternative to Least Squares Means</source>, 2nd Edn. New York, NY: Taylor &#x00026; Francis.</mixed-citation>
</ref>
<ref id="B49">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sokolova</surname> <given-names>M.</given-names></name> <name><surname>Lapalme</surname> <given-names>G.</given-names></name></person-group> (<year>2009</year>). <article-title>A systematic analysis of performance measures for classification tasks</article-title>. <source>Inf. Process. Manage</source>. <volume>45</volume>, <fpage>427</fpage>&#x02013;<lpage>437</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ipm.2009.03.002</pub-id></mixed-citation>
</ref>
<ref id="B50">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sousa-Lima</surname> <given-names>R.</given-names></name> <name><surname>Paglia</surname> <given-names>A.</given-names></name> <name><surname>Fonseca</surname> <given-names>G.</given-names></name></person-group> (<year>2008</year>). <article-title>Gender, age, and identity in the isolation calls of Antillean manatees (<italic>Trichechus manatus manatus</italic>)</article-title>. <source>Aquatic Mammals</source> <volume>34</volume>, <fpage>109</fpage>&#x02013;<lpage>122</lpage>. doi: <pub-id pub-id-type="doi">10.1578/AM.34.1.2008.109</pub-id></mixed-citation>
</ref>
<ref id="B51">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sousa-Lima</surname> <given-names>R. S.</given-names></name> <name><surname>Paglia</surname> <given-names>A. P.</given-names></name> <name><surname>Da Fonseca</surname> <given-names>G. A.</given-names></name></person-group> (<year>2002</year>). Signature information and individual recognition in the isolation calls of Amazonian manatees, <italic>Trichechus inunguis</italic> (mammalia: Sirenia). <italic>Anim. Behav</italic>. <volume>63</volume>, <fpage>301</fpage>&#x02013;<lpage>310</lpage>. doi: <pub-id pub-id-type="doi">10.1006/anbe.2001.1873</pub-id></mixed-citation>
</ref>
<ref id="B52">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stowell</surname> <given-names>D.</given-names></name> <name><surname>Wood</surname> <given-names>M. D.</given-names></name> <name><surname>Pamu&#x00142;a</surname> <given-names>H.</given-names></name> <name><surname>Stylianou</surname> <given-names>Y.</given-names></name> <name><surname>Glotin</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>Automatic acoustic detection of birds through deep learning: the first bird audio detection challenge</article-title>. <source>Methods Ecol. Evol</source>. <volume>10</volume>, <fpage>368</fpage>&#x02013;<lpage>380</lpage>. doi: <pub-id pub-id-type="doi">10.1111/2041-210X.13103</pub-id></mixed-citation>
</ref>
<ref id="B53">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tavabi</surname> <given-names>A.</given-names></name> <name><surname>Weninger</surname> <given-names>F.</given-names></name> <name><surname>Schuller</surname> <given-names>B.</given-names></name></person-group> (<year>2021</year>). &#x0201C;Automatic acoustic classification of feline sex from vocalizations," in <italic>Proceedings of the 29th ACM International Conference on Multimedia</italic> (New York, NY: ACM), <fpage>3765</fpage>&#x02013;<lpage>3769</lpage>.</mixed-citation>
</ref>
<ref id="B54">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Umeed</surname> <given-names>R.</given-names></name> <name><surname>Niemeyer Attademo</surname> <given-names>F. L.</given-names></name> <name><surname>Bezerra</surname> <given-names>B.</given-names></name></person-group> (<year>2018</year>). <article-title>The influence of age and sex on the vocal repertoire of the Antillean manatee (<italic>Trichechus manatus</italic> manatus) and their responses to call playback</article-title>. <source>Mar. Mamm. Sci</source>. <volume>34</volume>, <fpage>577</fpage>&#x02013;<lpage>594</lpage>. doi: <pub-id pub-id-type="doi">10.1111/mms.12467</pub-id></mixed-citation>
</ref>
<ref id="B55">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Varma</surname> <given-names>S.</given-names></name> <name><surname>Simon</surname> <given-names>R.</given-names></name></person-group> (<year>2006</year>). <article-title>Bias in error estimation when using cross-validation for model selection</article-title>. <source>BMC Bioinformatics</source> <volume>7</volume>:<fpage>91</fpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2105-7-91</pub-id><pub-id pub-id-type="pmid">16504092</pub-id></mixed-citation>
</ref>
<ref id="B56">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wierucka</surname> <given-names>K.</given-names></name> <name><surname>Budka</surname> <given-names>M.</given-names></name> <name><surname>Osiejuk</surname> <given-names>T. S.</given-names></name></person-group> (<year>2025</year>). <article-title>Same data, different results? Machine learning approaches in bioacoustics</article-title>. <source>Methods Ecol. Evol</source>. <volume>16</volume>:<fpage>70091</fpage>. doi: <pub-id pub-id-type="doi">10.1111/2041-210X.70091</pub-id></mixed-citation>
</ref>
<ref id="B57">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wrege</surname> <given-names>P. H.</given-names></name> <name><surname>Rowland</surname> <given-names>E. D.</given-names></name> <name><surname>Keen</surname> <given-names>S.</given-names></name> <name><surname>Shiu</surname> <given-names>Y.</given-names></name></person-group> (<year>2017</year>). <article-title>Acoustic monitoring for conservation in tropical forests: examples from forest elephants</article-title>. <source>Methods Ecol. Evol</source>. <volume>8</volume>, <fpage>1292</fpage>&#x02013;<lpage>1301</lpage>. doi: <pub-id pub-id-type="doi">10.1111/2041-210X.12730</pub-id></mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2240634/overview">Milan Tuba</ext-link>, Singidunum University, Serbia</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/42159/overview">Gilberto Corso</ext-link>, Federal University of Rio Grande do Norte, Brazil</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3133573/overview">Alastair Thomas Andrew Pickering</ext-link>, University College London, United Kingdom</p>
</fn>
</fn-group>
</back>
</article>