Psychiatric Voice Biomarkers: Methodological flaws in pediatric populations

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Introduction Psychiatric assessments rely on patient self-reports, clinician observations, and standardized scales, while objective technological tools are currently not reliable enough to be utilized in a clinical setting. Voice may be utilized as a biomarker in different scenarios, including differential diagnosis, assessing symptom severity and predicting suicidality. However, its use depends on accurate automatic speech recognition (ASR). Current gold standard open source ASR systems are trained mainly on adult speech and perform poorly in children, limiting application in pediatric psychiatry. Methods We benchmarked two open-source ASR models—NVIDIA Parakeet and Whisper-small—on the Ohio Child Speech Corpus (303 children, ages 4–9), using the reference human transcripts provided with the dataset. Audio was standardized to each model’s expected sampling rate. No model fine-tuning or adaptation was performed. For each utterance, we computed word error rate (WER) and character error rate (CER), and assessed semantic fidelity using Sentence Mover’s Distance (SMD) and BERTScore F1. Metrics were summarized overall, stratified by single-year age bins (4, 5, 6, 7, 8, 9), and also grouped into two broader categories: younger children (ages 4–6) and older children (ages 7–9). We compared WER, CER, SMD, and BERTScore F1 across both age groups and evaluated age effects as trends using nonparametric statistical tests. Results Both models showed significant age effects where younger children had markedly higher word error rates (WER >40%) and character error rates (CER >30%) compared to older children (WER ∼30%, CER ∼20%). Sentence mover distance improved with age, while BERTScore F1 remained stable. Despite age-related improvements, overall transcription accuracy was low. Discussion Current commonly used open-source ASR systems are inadequate for pediatric audio transcription, specifically in younger children. In order to build clinically translatable tools, collecting child-specific data and model fine-tuning through structured speech paradigms is essential.
Full text 35,987 characters · extracted from preprint-html · click to expand
Psychiatric Voice Biomarkers: Methodological flaws in pediatric populations | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Psychiatric Voice Biomarkers: Methodological flaws in pediatric populations View ORCID Profile Hammza Jabbar Abd Sattar Hamoudi , View ORCID Profile Mon-Ju Wu , View ORCID Profile Marsal Sanches , View ORCID Profile Cesar A. Soutullo , View ORCID Profile Carolina Olmos , View ORCID Profile Leslie K. Taylor , View ORCID Profile Giovanna Zunta-Soares , View ORCID Profile Jair C. Soares , View ORCID Profile Benson Mwangi doi: https://doi.org/10.1101/2025.10.13.25337901 Hammza Jabbar Abd Sattar Hamoudi 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Hammza Jabbar Abd Sattar Hamoudi For correspondence: hammza.jabbarabdlsattarhamoudi{at}uth.tmc.edu Mon-Ju Wu 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Mon-Ju Wu Marsal Sanches 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Marsal Sanches Cesar A. Soutullo 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Cesar A. Soutullo Carolina Olmos 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Carolina Olmos Leslie K. Taylor 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Leslie K. Taylor Giovanna Zunta-Soares 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Giovanna Zunta-Soares Jair C. Soares 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jair C. Soares Benson Mwangi 1 UT Center of Excellence on Mood Disorders, School of Behavioral Health Sciences, The University of Texas Health Science Center at Houston , Houston, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Benson Mwangi Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Introduction Psychiatric assessments rely on patient self-reports, clinician observations, and standardized scales, while objective technological tools are currently not reliable enough to be utilized in a clinical setting. Voice may be utilized as a biomarker in different scenarios, including differential diagnosis, assessing symptom severity and predicting suicidality. However, its use depends on accurate automatic speech recognition (ASR). Current gold standard open source ASR systems are trained mainly on adult speech and perform poorly in children, limiting application in pediatric psychiatry. Methods We benchmarked two open-source ASR models—NVIDIA Parakeet and Whisper-small—on the Ohio Child Speech Corpus (303 children, ages 4–9), using the reference human transcripts provided with the dataset. Audio was standardized to each model’s expected sampling rate. No model fine-tuning or adaptation was performed. For each utterance, we computed word error rate (WER) and character error rate (CER), and assessed semantic fidelity using Sentence Mover’s Distance (SMD) and BERTScore F1. Metrics were summarized overall, stratified by single-year age bins (4, 5, 6, 7, 8, 9), and also grouped into two broader categories: younger children (ages 4–6) and older children (ages 7–9). We compared WER, CER, SMD, and BERTScore F1 across both age groups and evaluated age effects as trends using nonparametric statistical tests. Results Both models showed significant age effects where younger children had markedly higher word error rates (WER >40%) and character error rates (CER >30%) compared to older children (WER ∼30%, CER ∼20%). Sentence mover distance improved with age, while BERTScore F1 remained stable. Despite age-related improvements, overall transcription accuracy was low. Discussion Current commonly used open-source ASR systems are inadequate for pediatric audio transcription, specifically in younger children. In order to build clinically translatable tools, collecting child-specific data and model fine-tuning through structured speech paradigms is essential. Introduction Audio Biomarkers for Psychiatric Disorders Psychiatric assessments are primarily based on patient subjective reports, clinician observations during the mental status examination that integrate both objective signs and subjective impressions, and the use of standardized clinician-rated and patient self-reported scales to quantify symptom severity and possibly establish a diagnosis or guide possible changes in diagnosis over the course of illness ( 1 ). Despite advances in neuroscience and digital health, psychiatric assessments still lack objective technological tools for diagnosis and continue to rely on traditional, subjective methods ( 2 ). Voice biomarkers for mental health remain an underexplored field with a great potential for diagnostic differentiation. For instance, a cross-sectional study observing voice biomarkers in adults with major depressive disorder (MDD), demonstrated that a Machine Learning, Support Vector Machine (SVM) Classifier trained on acoustic and linguistic features such as fundamental frequency, jitter, shimmer, pause structure, and prosodic variability was able to distinguish MDD from healthy controls with an area under curve (AUC) of 0.93( 3 ). However, differentiating between overlapping psychiatric conditions (e.g., Unipolar Depression vs Bipolar Depression) remains a challenge as compared to differentiating against healthy controls. Cross-diagnostic classification studies often achieve only modest accuracies in the 60–70% range, mainly due to limited datasets, and scarcity of standardized speech paradigm protocols( 4 ). Current transcription models and lack of pediatric models Psychiatric audio/speech biomarker studies, including those exploring depression, bipolar disorder, and suicide risk, are dependent on the quality of the underlying transcription models used to process speech recordings. These models belong to the broader class of automatic speech recognition (ASR) systems, which map acoustic waveforms into text representations using deep learning architectures such as recurrent neural networks, transformers, or hybrid approaches( 5 ). Modern foundation-scale ASR systems trained on thousands of hours of speech have achieved impressive robustness to noise, accents, and domain shifts, and can generalize to unseen tasks through zero-shot inference( 6 ). Despite these advances, ASR systems are mainly trained on adult speech, and their performance drops substantially in children( 7 ). Younger children in particular present challenges due to higher fundamental frequency, shorter vocal tract length, immature articulation, variable pronunciation, and frequent disfluencies( 8 ). A key solution to this technological gap is fine-tuning models with child-specific data, specifically data from healthy children and children with psychiatric conditions. That is, using this data to train the models, so that their accuracy becomes sufficient for a clinically translatable tool for differential diagnostic decision-making. Therefore, to establish ASR model accuracy we benchmarked two models - NVIDIA Parakeet and Whisper small model against professionally done human transcription benchmarks. In both models, we evaluated transcription accuracy using the Ohio Child Speech Corpus (OCSC)( 9 ), an open-access dataset of 303 children aged 4–9 recorded in a public museum lab setting. OCSC offers clean lapel-mic audio, diverse structured and spontaneous tasks (e.g., letter-naming, story descriptions, Wug test), and detailed metadata on age, sex, and socio-environmental context. Data were acquired through the Talkbank speech data repository( 10 ). We selected OCSC due to its high-quality child-directed audio, rich annotations, and public availability—making it ideally suited for benchmarking pediatric ASR models and guiding future adaptation for clinical use. Methods Audio data from the Ohio Child Speech Corpus was acquired from talkbank.org and subsequently processed using a custom Python pipeline that leverages CHAT/CHILDES transcript files (.cha)( 11 ) to perform speaker-aware ASR evaluation. The pipeline parsed .cha files using the pylangacq library( 11 ) to extract utterance-level transcripts, speaker identities, and precise timing information (start/end timestamps in milliseconds). For each utterance, the corresponding audio segment was extracted from MP3 recordings using pydub( 12 ), resampled to 16kHz mono, and submitted individually to one of two ASR engines, NVIDIA Parakeet or Whisper. This utterance-level approach eliminated the need for automatic speaker diarization, as ground-truth speaker labels and timing were directly available from the .cha metadata. We benchmarked NVIDIA Parakeet - TDT - 0.6B (v2) and OpenAI Whisper - small - both compact yet capable models selected for their recent design and manageable inference footprint. Parakeet-TDT-0.6B is a ∼600 million-parameter FastConformer-TDT model (released May 2025) downloadable from Hugging Face. The Parakeet model uses a transducer-style decoder that jointly models tokens and their durations, allowing efficient skipping of silent frames. On the other hand, whisper-small (∼244 million parameters) is a Transformer encoder–decoder model made available through the OpenAI/Hugging Face repository and was trained on ∼680,000 hours of (weakly supervised) speech with strong generalization across domains. Both models report word error rates in the single-digit (∼ 6-9 %) based on standard adult English benchmarks as highlighted by the huggingface Opean ASR leaderboad ( 13 ) Noticeably, as both models are small, they offer lower latency and memory use - important for real-time or edge deployment - while still achieving competitive WER. Our experiments ran on a single 48 GB NVIDIA RTX 6000 Ada GPU. Statistical analyses were performed using traditional metrics including Word Error Rate (WER), Character Error Rate (CER), using the jiwer library( 14 ). Semantic similarity metrics BERTScore( 15 ) F1 and Sentence Mover’s Distance - were calculated using the Bertscore( 15 ) and the sentence transformers library( 16 ) respectively. Statistical analyses and visualization were performed using Python with scipy stats, seaborn, and matplotlib libraries. Participant-level ASR metrics were compared across age groups using one-way ANOVA with Tukey’s HSD post-hoc tests, independent samples t-tests for young (ages 4-6) vs. old (ages 7-9) comparisons, and correlation analyses with effect size calculations (Cohen’s d). Results Dataset Characteristics The dataset consisted of 303 participants, with an average audio duration of 31.2 minutes per child (range: 9.2–73.5 minutes). Younger children (ages 4–6; n = 139) and older children (ages 7–9; n = 164) contributed to the analyses. Nvidia Parakeet Group-Level Comparisons: Younger vs. Older Children Significant differences in transcription performance were observed across the two age groups. Word Error Rate (WER) was higher in the younger group (mean = 46.5%, SD = 16.9) compared to the older group (mean = 32.2%, SD = 14.4), t(301) = 7.87, p < 0.0001, with a large effect size (Cohen’s d = 0.92). Character Error Rate (CER) followed the same pattern, with younger children showing a mean of 31.3% (SD = 12.4) compared to 21.5% (SD = 10.9) in older children, t(301) = 7.26, p < 0.0001, Cohen’s d = 0.85. For semantic similarity metrics, Sentence Mover’s Distance (SMD) was significantly lower in older children (mean = 0.15, SD = 0.07) relative to younger children (mean = 0.18, SD = 0.07), t(301) = 3.38, p = 0.0008, Cohen’s d = 0.39. In contrast, BERTScore F1 did not differ significantly between groups, with values of –0.05 (SD = 0.12) for younger children and –0.04 (SD = 0.12) for older children, t(301) = –1.10, p = 0.27, Cohen’s d = –0.13. Utterance count was also comparable between groups, averaging 837 (SD = 261) in the younger group and 876 (SD = 269) in the older group, t(301) = –1.26, p = 0.21, Cohen’s d = –0.14 ( Table 1 ). View this table: View inline View popup Download powerpoint Table 1. Group comparisons of transcription performance metrics from the Parakeet model between younger and older participants. Age-Specific Effects (4–9 Years) One-way ANOVAs were conducted to examine transcription performance across individual ages from 4 to 9 years. For WER ( Figure 1-A ), there was a significant main effect of age (p < 0.0001). Error rates were highest at age 4 (median values above 50%) and progressively declined through ages 5 and 6, reaching the lowest levels at ages 8 and 9 (median values near 25%). Post-hoc comparisons revealed that ages 8 and 9 had significantly lower WER compared to ages 4 through 6 (all p < 0.01). Download figure Open in new tab Figure 1. Distribution of automatic speech recognition (ASR) metrics from the Nvidia Parakeet model by age. Each boxplot represents the distribution across participants aged 4–9 years for (A) BERTScore-F1, (B) Character Error Rate (CER), (C) Sentence Mover Distance, and (D) Word Error Rate (WER). CER ( Figure 1-B ) also showed a significant effect of age (p < 0.0001). Children aged 4 displayed mean CER values above 35–40%, while those aged 8 and 9 averaged around 15–20%. Post-hoc testing confirmed significant differences between ages 8–9 and ages 4–6, with older children consistently achieving lower error rates. For semantic similarity, SMD ( Figure 1-C ) varied significantly across ages (p = 0.0001). Median values were highest in the youngest group (age 4, approximately 0.23) and lowest in the oldest group (age 9, approximately 0.12–0.13). Post-hoc comparisons showed significant reductions between age 4 and ages 8–9, suggesting gradual improvements in semantic fidelity with age. In contrast, BERTScore F1 ( Figure 1-D ) did not differ significantly across ages (p = 0.16). Scores remained relatively stable, with median values centered around –0.05 to –0.02 across all ages, and no clear monotonic trend. Correlation Analyses Correlation analyses ( Figure 2 ) demonstrated strong relationships among performance metrics. WER and CER were highly correlated (r > 0.8, p < 0.001), confirming that both measures captured overlapping variance in transcription accuracy. Both WER and CER were negatively correlated with BERTScore F1, indicating that lower error rates were associated with improved embedding-based similarity. Conversely, WER and CER were positively correlated with SMD, showing that higher error rates corresponded to greater semantic distortion in the transcriptions. These associations underscore the consistency across error-based and semantic measures. Download figure Open in new tab Figure 2. Correlation matrix of Nvidia Parakeet model performance metrics and age. Positive correlations are shown in red and negative correlations in blue. Results – OpenAI Whisper Small Model Group-Level Comparisons: Younger vs. Older Children Significant differences in transcription performance were observed across the two age groups. Word Error Rate (WER) was higher in the younger group (mean = 41.3%, SD = 15.1) compared to the older group (mean = 29.0%, SD = 12.7), t (301) = 7.58, p < 0.0001, with a large effect size (Cohen’s d = 0.89). Character Error Rate (CER) followed the same pattern, with younger children showing a mean of 29.8% (SD = 16.5) compared to 20.3% (SD = 10.0) in older children, t (301) = 5.91, p < 0.0001, Cohen’s d = 0.71. For semantic similarity, Sentence Mover’s Distance (SMD) was significantly lower in older children (mean = 0.14, SD = 0.07) relative to younger children (mean = 0.16, SD = 0.06), t (301) = 2.61, p = 0.0096, Cohen’s d = 0.30. In contrast, BERTScore F1 did not differ significantly between groups, with values of –0.0004 (SD = 0.12) for younger children and –0.0036 (SD = 0.11) for older children, t (301) = 0.24, p = 0.81, Cohen’s d = 0.03. Utterance count was also comparable between groups, averaging 823 (SD = 228) in the younger group and 847 (SD = 201) in the older group, t (301) = –0.95, p = 0.34, Cohen’s d = –0.11 ( Table 2 ). View this table: View inline View popup Table 2. Group comparisons of transcription performance metrics from the OpenAI Whisper Small model between younger and older participants. Age-Specific Effects (4–9 Years) One-way ANOVAs were conducted to examine transcription performance across individual ages from 4 to 9 years. For WER ( Figure 3-A ), there was a significant main effect of age ( p 50%) and progressively declined through ages 5 and 6, reaching the lowest levels at ages 8 and 9 (median values near 25%). Post-hoc comparisons revealed that ages 8 and 9 had significantly lower WER compared to ages 4 through 6 (all p < 0.01). Download figure Open in new tab Figure 3. Distribution of automatic speech recognition (ASR) metrics from the OpenAI Whisper Small model by age. Each boxplot represents the distribution across participants aged 4–9 years for (A) BERTScore-F1, (B) Character Error Rate (CER), (C) Sentence Mover Distance, and (D) Word Error Rate (WER). CER ( Figure 3-B ) also showed a significant effect of age ( p < 0.0001). Children aged 4 displayed mean CER values above 35–40%, while those aged 8 and 9 averaged around 15–20%. Post-hoc testing confirmed significant differences between ages 8–9 and ages 4–6, with older children consistently achieving lower error rates. For semantic similarity ( Figure 3-C ), SMD varied significantly across ages ( p < 0.0001). Median values were highest in the youngest group (age 4, approximately 0.22–0.23) and lowest in the oldest group (age 9, approximately 0.12–0.13). Post-hoc comparisons showed significant reductions between age 4 and ages 8–9, suggesting gradual improvements in semantic fidelity with age. In contrast, BERTScore F1 ( Figure 3-D ) did not differ significantly across ages ( p = 0.16). Scores remained relatively stable, with median values centered around –0.05 to –0.02 across all ages, and no clear monotonic trend. Correlation Analyses Correlation analyses ( Figure 4 ) demonstrated strong relationships among performance metrics. WER and CER were highly correlated (r > 0.98, p < 0.001), confirming that both measures captured overlapping variance in transcription accuracy. Both WER and CER were negatively correlated with BERTScore F1, indicating that lower error rates were associated with improved embedding-based similarity. Conversely, both error metrics were positively correlated with SMD, showing that higher error rates corresponded to greater semantic distortion. Age was negatively correlated with both WER and CER, further confirming that transcription accuracy improved with development. Download figure Open in new tab Figure 4. Correlation matrix of OpenAI Whisper Small model performance metrics and age. Positive correlations are shown in red and negative correlations in blue. Discussion The NVIDIA Parakeet model showed large developmental effects but overall high transcription error rates across the pediatric sample. In group-level comparisons, WER averaged 46.5% (SD = 16.9) in younger children (ages 4–6) and 32.2% (SD = 14.4) in older children (ages 7–9). CER values followed the same pattern, decreasing from 31.3% (SD = 12.4) in the younger group to 21.5% (SD = 10.9) in the older group. Semantic fidelity, indexed by Sentence Mover’s Distance, improved modestly with age (0.18 vs. 0.15). However, these error rates remain high compared to adult benchmarks, and BERTScore F1 did not show meaningful differences across groups. ANOVA confirmed significant main effects of age for WER, CER, and SMD, with the steepest contrasts between ages 8–9 and 4–6. Correlations confirmed that higher error rates were tightly coupled with lower semantic fidelity. The Whisper Small model revealed a nearly identical developmental trajectory, with even younger children experiencing error rates exceeding 40% WER and 30% CER. Older children performed somewhat better, with WER values in the low 30% range and CER around the low 20s, but performance was still far below levels required for reliable clinical transcription. SMD decreased with age, indicating improved semantic similarity, but absolute values remained poor compared to adult-level ASR. As with Parakeet, BERTScore F1 did not differ across groups, and utterance counts were stable. Age-stratified analyses showed significant declines in error measures from age 4 to ages 8–9, but the absolute performance at all ages was far worse than typical adult speech recognition. Taken together, these findings highlight the current technological limitation: neither Parakeet nor Whisper Small provide adequate transcription performance for pediatric speech, particularly for younger children. Both models demonstrate systematic age effects, but even at ages 8–9, error rates remain too high for dependable downstream use in psychiatric audio biomarker research. The extremely poor performance in the youngest children underscores that current open-source ASR systems are not trained for child speech and cannot yet serve as a foundation for pediatric psychiatry applications without substantial adaptation or fine-tuning. These results collectively point to a clear gap in the field: there are currently no robust, validated open-source transcription models for pediatric psychiatry, and meaningful clinical progress will require dedicated pediatric data and fine-tuning approaches. The development of clinically translatable pediatric ASR systems would open pathways to the development of translational differential diagnostic and monitoring tools in psychiatry. Voice features have already been linked to symptom severity and treatment response in adult depression and bipolar disorder( 17 ), and extending these approaches to pediatric populations could support earlier identification and longitudinal tracking. A particularly promising avenue is distinguishing pediatric unipolar versus bipolar depression( 18 ), a clinically challenging differential diagnosis where early, objective tools may improve treatment allocation. Beyond mood disorders, standardized voice tasks may help differentiate autism spectrum disorder (ASD) and ADHD( 19 ), or aid in parsing overlapping neurodevelopmental and psychiatric conditions. Data Availability All data produced in the present work are contained in the manuscript Conflict of Interest The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest. Acknowledgments This work was supported by the NECMHR01 grant from the Texas Child Mental Health Care Consortium (TCMHCC) and the Dunn Foundation. Speech dataset acquired from the Ohio Child Speech Corpus (OCSC) through talkbank.org References 1. ↵ Newson JJ , Hunter D , Thiagarajan TC . The Heterogeneity of Mental Health Assessment . Front Psychiatry [Internet] . 2020 Feb 27 [cited 2025 Oct 1 ]; 11 . Available from: https://www.frontiersin.org/journals/psychiatry/articles/10.3389/fpsyt.2020.00076/full 2. ↵ Insel TR . Digital Phenotyping: Technology for a New Science of Behavior . JAMA . 2017 Oct 3; 318 ( 13 ): 1215 – 6 . OpenUrl CrossRef PubMed 3. ↵ ResearchGate [Internet] . [cited 2025 Oct 1 ]. The voice of depression: speech features as biomarkers for major depressive disorder . Available from: https://www.researchgate.net/publication/385750565_The_voice_of_depression_speech_fe atures_as_biomarkers_for_major_depressive_disorder 4. ↵ Low DM , Bentley KH , Ghosh SS . Automated assessment of psychiatric disorders using speech: A systematic review . Laryngoscope Investig Otolaryngol . 2020 Jan 31; 5 ( 1 ): 96 – 116 . OpenUrl CrossRef PubMed 5. ↵ Radford A , Kim JW , Xu T , Brockman G , McLeavey C , Sutskever I. Robust Speech Recognition via Large-Scale Weak Supervision [Internet]. arXiv ; 2022 [cited 2025 Oct 1 ]. Available from: http://arxiv.org/abs/2212.04356 6. ↵ Larochelle H , Ranzato M , Hadsell R , Balcan MF , Lin H , editors Baevski A , Zhou Y , Mohamed A , Auli M. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations . In: Larochelle H , Ranzato M , Hadsell R , Balcan MF , Lin H , editors. Advances in Neural Information Processing Systems [Internet] . Curran Associates, Inc .; 2020 . p. 12449 – 60 . Available from: https://proceedings.neurips.cc/paper_files/paper/2020/file/92d1e1eb1cd6f9fba3227870bb6d7f07-Paper.pdf 7. ↵ GerosaMatteo, GiulianiDiego, BrugnaraFabio . Acoustic variability and automatic recognition of children’s speech. Speech Commun [Internet] . 2007 Oct 1 [cited 2025 Oct 1 ]; Available from: https://dl.acm.org/doi/10.1016/j.specom.2007.01.002 8. ↵ Bhardwaj V , Ben Othman MT , Kukreja V , Belkhier Y , Bajaj M , Goud BS , et al. Automatic Speech Recognition (ASR) Systems for Children: A Systematic Literature Review . Appl Sci . 2022 Jan ; 12 ( 9 ): 4419 . OpenUrl 9. ↵ Wagner L , Alghowinhem S , Alwan A , Bowdrie K , Breazeal C , Clopper CG , et al. The Ohio Child Speech Corpus . Speech Commun . 2025 May ; 170 : 103206 . OpenUrl 10. ↵ MacWhinney B. The talkbank project. In Creating and digitizing language corpora: Volume 1: Synchronic databases . In: The talkbank project In Creating and digitizing language corpora: Volume 1: Synchronic databases . London : Palgrave Macmillan UK ; p. 163 – 80 . 11. ↵ ResearchGate [Internet] . [cited 2025 Oct 9 ]. The CHILDES Project: Tools for Analyzing Talk (third edition): Volume I: Transcription format and programs, Volume II: The database . Available from: https://www.researchgate.net/publication/238833352_The_CHILDES_Project_Tools_for_Analyzing_Talk_third_edition_Volume_I_Transcription_format_and_programs_Volume_II_The_database 12. ↵ Robert J. jiaaro/pydub [Internet] . 2025 [cited 2025 Oct 9 ]. Available from: https://github.com/jiaaro/pydub 13. ↵ Open ASR Leaderboard - a Hugging Face Space by hf-audio [Internet] . [cited 2025 Oct 9 ]. Available from: https://huggingface.co/spaces/hf-audio/open_asr_leaderboard 14. ↵ jiwer: Evaluate your speech-to-text system with similarity measures such as word error rate (WER) . 15. ↵ bert-score: PyTorch implementation of BERT score [Internet] . [cited 2025 Oct 9 ]. Available from: https://github.com/Tiiiger/bert_score 16. ↵ sentence-transformers: Embeddings, Retrieval, and Reranking [Internet] . [cited 2025 Oct 9 ]. Available from: https://www.SBERT.net 17. ↵ Cummins N , Scherer S , Krajewski J , Schnieder S , Epps J , Quatieri TF . A review of depression and suicide risk assessment using speech analysis . Speech Commun . 2015 July 1; 71 : 10 – 49 . OpenUrl CrossRef 18. ↵ Zhong R , Wu X , Chen J , Fang Y. Using Digital Phenotyping to Discriminate Unipolar Depression and Bipolar Disorder: Systematic Review . J Med Internet Res . 2025 May 23; 27 ( 1 ): e72229 . OpenUrl PubMed 19. ↵ “Is voice a marker for Autism spectrum disorder? A systematic review and meta-analysis” - Fusaroli - 2017 - Autism Research - Wiley Online Library [Internet] . [cited 2025 Oct 1 ]. Available from: https://onlinelibrary.wiley.com/doi/10.1002/aur.1678 View the discussion thread. Back to top Previous Next Posted October 15, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Psychiatric Voice Biomarkers: Methodological flaws in pediatric populations Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Psychiatric Voice Biomarkers: Methodological flaws in pediatric populations Hammza Jabbar Abd Sattar Hamoudi , Mon-Ju Wu , Marsal Sanches , Cesar A. Soutullo , Carolina Olmos , Leslie K. Taylor , Giovanna Zunta-Soares , Jair C. Soares , Benson Mwangi medRxiv 2025.10.13.25337901; doi: https://doi.org/10.1101/2025.10.13.25337901 Share This Article: Copy Citation Tools Psychiatric Voice Biomarkers: Methodological flaws in pediatric populations Hammza Jabbar Abd Sattar Hamoudi , Mon-Ju Wu , Marsal Sanches , Cesar A. Soutullo , Carolina Olmos , Leslie K. Taylor , Giovanna Zunta-Soares , Jair C. Soares , Benson Mwangi medRxiv 2025.10.13.25337901; doi: https://doi.org/10.1101/2025.10.13.25337901 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Psychiatry and Clinical Psychology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15228) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6598) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (540) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9231) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00538f12927dde1',t:'MTc3OTU1MTQ5MA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00