<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Form Res</journal-id><journal-id journal-id-type="publisher-id">formative</journal-id><journal-id journal-id-type="index">27</journal-id><journal-title>JMIR Formative Research</journal-title><abbrev-journal-title>JMIR Form Res</abbrev-journal-title><issn pub-type="epub">2561-326X</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v10i1e92646</article-id><article-id pub-id-type="doi">10.2196/92646</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>A Smartphone-Based Acoustic Machine Learning Pipeline for Detecting Suicidal Ideation: Case-Control Model Development and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Lyu</surname><given-names>Min</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Tan</surname><given-names>Lixin</given-names></name><degrees>MEd</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Xiao</surname><given-names>Jun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Huang</surname><given-names>Heqing</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Fangjian</given-names></name><degrees>MAP</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Qi</surname><given-names>Jiahui</given-names></name><degrees>MEd</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Huang</surname><given-names>Tian</given-names></name><degrees>MAP</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Lei</surname><given-names>Jinyu</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Zhihui</given-names></name><degrees>MEng</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jiang</surname><given-names>Tianxiang</given-names></name><degrees>MEd</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Zhu</given-names></name><degrees>MAP</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Xueqian</given-names></name><degrees>MAP</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhong</surname><given-names>Jiang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Feng</surname><given-names>Zhengzhi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff7">7</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Medical Psychology, Army Medical University</institution><addr-line>No. 30 Gaotanyan Main Street, Shapingba District</addr-line><addr-line>Chongqing</addr-line><country>China</country></aff><aff id="aff2"><institution>Mental Health Education and Counseling Center, Chongqing University</institution><addr-line>Chongqing</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Cardiovascular Medicine, Chongqing Emergency Medical Center/The Fourth People's Hospital of Chongqing</institution><addr-line>Chongqing</addr-line><country>China</country></aff><aff id="aff4"><institution>Department of Medical Psychology, First Affiliated Hospital of Army Medical University</institution><addr-line>Chongqing</addr-line><country>China</country></aff><aff id="aff5"><institution>Faculty of Health and Wellness, City University of Macau</institution><addr-line>Macau</addr-line><country>China</country></aff><aff id="aff6"><institution>College of Computer Science, Chongqing University</institution><addr-line>Chongqing</addr-line><country>China</country></aff><aff id="aff7"><institution>College of Medicine, Chongqing University</institution><addr-line>Chongqing</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Garcia-Montalvo</surname><given-names>Ivan Antonio</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ashraf</surname><given-names>Walid</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Zhengzhi Feng, PhD, Department of Medical Psychology, Army Medical University, No. 30 Gaotanyan Main Street, Shapingba District, Chongqing, 400038, China, +86-16623460852; <email>fzz@tmmu.edu.cn</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>15</day><month>9</month><year>2026</year></pub-date><volume>10</volume><elocation-id>e92646</elocation-id><history><date date-type="received"><day>01</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>09</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>29</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Min Lyu, Lixin Tan, Jun Xiao, Heqing Huang, Fangjian Liu, Jiahui Qi, Tian Huang, Jinyu Lei, Zhihui Zhao, Tianxiang Jiang, Zhu Liu, Xueqian Wang, Jiang Zhong, Zhengzhi Feng. Originally published in JMIR Formative Research (<ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>), 15.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Formative Research, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://formative.jmir.org">https://formative.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://formative.jmir.org/2026/1/e92646"/><abstract><sec><title>Background</title><p>Suicidal ideation (SI) among university students is a growing public health concern. Self-report screening can be limited by concealment and delayed disclosure. We evaluated a leakage-resistant, proof-of-concept pipeline to detect SI from standardized smartphone-recorded speech.</p></sec><sec><title>Objective</title><p>This study aimed to extract acoustic markers from brief smartphone-based reading tasks and develop machine learning models for suicide risk prediction in university students, enabling low-cost, scalable early screening to support campus mental health services.</p></sec><sec sec-type="methods"><title>Methods</title><p>Questionnaire data and speech recordings were collected via a WeChat mini program. After screening and clinical confirmation, 96 participants (n=48, 50% with SI; n=48, 50% controls) were included. Age and sex were evaluated as potential confounders. Each participant read 16 standardized sentences. Acoustic features were extracted using openSMILE (version 3.0.2), yielding a 570D feature vector per utterance. To prevent leakage from multiple recordings per speaker, we used participant-level 5-fold cross-validation, assigning all recordings from each participant to a single fold. Seven machine learning algorithms were evaluated using area under the curve (AUC), accuracy, and <italic>F</italic><sub>1</sub>-score.</p></sec><sec sec-type="results"><title>Results</title><p>Acoustic-based models discriminated participants with SI from control participants. The SI group was significantly older than the control group (<italic>P</italic>=.001). Random forest achieved an AUC of 0.813 (accuracy=0.748), and naive Bayes achieved an AUC of 0.806 (accuracy=0.757). Feature families related to pitch, mel-frequency cepstral coefficients, and harmonicity contributed to model performance.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Standardized read speech captured via smartphones shows preliminary feasibility for SI discrimination under a leakage-aware evaluation design. External validation and testing with more naturalistic speech are warranted.</p></sec><sec><title>Trial Registration</title><p>Chinese Clinical Trial Registry ChiCTR2500106625; https://www.chictr.org.cn/showproj.html?proj=276927</p></sec></abstract><kwd-group><kwd>suicidal ideation</kwd><kwd>speech acoustics</kwd><kwd>machine learning</kwd><kwd>university students</kwd><kwd>cross-validation</kwd><kwd>artificial intelligence</kwd><kwd>AI</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Suicide is recognized by the World Health Organization as a major public health concern, underscoring the importance of early detection and timely intervention [<xref ref-type="bibr" rid="ref1">1</xref>]. Despite advances in mental health services, conventional suicide risk assessment&#x2014;largely reliant on self-report scales and clinical interviews&#x2014;has well-documented limitations, including susceptibility to social desirability and disclosure bias, limited sensitivity to rapid fluctuations in suicidal ideation (SI), and poor scalability in resource-constrained university settings [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. These challenges motivate the development of objective, noninvasive indicators that could support continuous and large-scale screening.</p><p>University students, as emerging young adults, face academic demands, interpersonal adjustment, and career-related pressures that may heighten psychological distress and vulnerability to suicidal thoughts and behaviors [<xref ref-type="bibr" rid="ref4">4</xref>]. Epidemiological studies report relatively high prevalence of SI in university samples [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref9">9</xref>]. Moreover, limited social support and restricted access to mental health services have been identified as important contributors to risk in this population [<xref ref-type="bibr" rid="ref10">10</xref>-<xref ref-type="bibr" rid="ref12">12</xref>]. Beyond prevalence estimates, recent work highlights the interplay of psychological, social, and cultural influences on suicidality, informing early identification and campus-based intervention strategies [<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref14">14</xref>]. However, objective and efficient risk assessment approaches suitable for real-world university contexts remain necessary.</p><p>In acoustic analysis, there is accumulating evidence suggesting that speech may provide objective and scalable signals associated with suicidality and suicide risk. In prior studies, individuals at elevated suicide risk have been reported to exhibit atypical vocal patterns, including increased fundamental frequency (F0) variability, higher jitter and shimmer, altered mel-frequency cepstral coefficient (MFCC) profiles, and differences in spectral energy distribution [<xref ref-type="bibr" rid="ref15">15</xref>,<xref ref-type="bibr" rid="ref16">16</xref>]. These acoustic deviations are plausibly linked to underlying mechanisms implicated in suicidality, such as heightened emotional dysregulation, physiological stress responses, and increased cognitive load [<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. Consistent with this literature, a systematic review of acoustic machine learning studies identified jitter, F0, MFCCs, and power spectral density as among the most frequently used and potentially discriminative features while emphasizing substantial methodological heterogeneity (eg, recording tasks and feature extraction pipelines) that may limit generalizability [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>With advances in AI and machine learning, an increasing number of studies have leveraged acoustic features (eg, MFCCs, formants, jitter, and energy) to predict suicide risk. Recent studies have reported promising results using diverse speech data and modeling approaches. Min et al [<xref ref-type="bibr" rid="ref21">21</xref>] achieved 79% accuracy for within-person prediction of worsening suicidality and 69% accuracy for between-person classification. Su et al [<xref ref-type="bibr" rid="ref22">22</xref>] achieved 75% accuracy using audio segments from crisis hotline calls. Krautz et al [<xref ref-type="bibr" rid="ref23">23</xref>] analyzed the voice recordings of individuals who completed suicide and achieved 76% accuracy. Ding et al [<xref ref-type="bibr" rid="ref24">24</xref>] constructed a crisis hotline suicide risk speech dataset and achieved 96% accuracy using a deep learning model. Lin et al [<xref ref-type="bibr" rid="ref18">18</xref>] achieved 77.82% accuracy combining acoustic and word frequency features under negative emotional conditions. Zhu et al [<xref ref-type="bibr" rid="ref25">25</xref>] achieved an area under the curve (AUC) of up to 1.00 for SI classification. Despite these encouraging results, evidence in university student populations remains limited, and variability in recording contexts and analytic pipelines continues to challenge robust and transferable models [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Therefore, we designed a study focusing on university students to extract key acoustic markers from brief smartphone-based reading tasks and develop suicide risk prediction models using machine learning algorithms. By using a low-cost and easily deployable recording protocol, this approach aims to support scalable early screening that could be integrated into campus mental health services to facilitate timely, data-informed prevention and intervention.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><p>We conducted a two-stage method development study to (1) develop and validate a set of standardized sentences for Chinese university students and (2) develop an acoustic-based machine learning pipeline for detecting SI. Importantly, the sentence validation cohort and the model development cohort were mutually exclusive to ensure independent evaluation across stages and prevent data leakage.</p><sec id="s2-1"><title>Ethical Considerations</title><p>This study received approval from the ethics committee of the Fourth People&#x2019;s Hospital of Chongqing (approval 2025; ethical review 73). Informed consent was obtained from all participants and/or their parents or legal guardians, in line with the Declaration of Helsinki. Participants with SI were further evaluated and intervened accordingly.</p></sec><sec id="s2-2"><title>Stage 1: Corpus Validation</title><p>A total of 54 university students aged 18 to 25 years were recruited via an online survey platform to rate candidate sentences. These participants did not take part in stage 2. A pool of 50 candidate sentences was created to cover three domains: (1) positive-valence statements, (2) negative-valence expressions, and (3) neutral factual statements. The positive-valence statements were imbued with Chinese traditional virtues and socialist core values to strengthen cultural resonance. To elicit diverse prosodic patterns, sentences were varied across syntactic moods (declarative, interrogative, and exclamatory). Sentence length was restricted to 100 Chinese characters or less, with a reading duration of less than 30 seconds to maintain feasibility and reduce fatigue.</p><p>Participants rated all 50 sentences on 9-point Likert scales for pleasure, arousal, and dominance, generating quantitative affective profiles. Two licensed clinical psychologists independently evaluated the sentences for readability, emotional expressiveness, and categorical balance. On the basis of consensus, 12 sentences (4 per valence domain) were selected. Four suicide-related statements adapted from validated instruments [<xref ref-type="bibr" rid="ref26">26</xref>] were then added, resulting in a final set of 16 standardized prompts (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s2-3"><title>Stage 2: Model Development</title><p>We conducted a case-control study among university students in southwest China using a sequential screening-and-confirmation process. Data were collected via a WeChat mini program, yielding 747 questionnaire responses and 771 speech recording submissions. Speech recordings were collected on participants&#x2019; personal smartphones in familiar, low-stress environments to enhance comfort and ecological validity [<xref ref-type="bibr" rid="ref27">27</xref>]. Raw audio files were stored in encrypted, access-controlled servers and were used only for acoustic feature extraction [<xref ref-type="bibr" rid="ref28">28</xref>]. After excluding incomplete entries and screening for audio quality, 715 participants remained eligible for SI screening. SI was initially screened using the Self-Rating Idea of Suicide Scale (SIOSS; threshold &#x2265;12). To reduce misclassification due to concealment, individuals scoring 4 or more on the SIOSS concealment subscale were excluded. Participants who were screened as positive then underwent counselor-administered clinical confirmation using the Columbia-Suicide Severity Rating Scale, requiring an ideation subscale score of 2 or more. This process identified 48 clinically confirmed SI cases. From those who were screened as negative on the SIOSS (&#x003C;12), we randomly selected 48 controls (1:1) using a computer-generated random sample. Age and sex were compared between groups and were adjusted for in subsequent analyses. The final analytic sample comprised 96 participants. Each participant recorded 16 standardized utterances in quiet settings, yielding 1536 recordings for model development.</p></sec><sec id="s2-4"><title>Audio Preprocessing and Reading Compliance Checking</title><p>All recordings underwent a standardized preprocessing pipeline. Audio was first enhanced using the Model Scope Acoustic Noise Suppression model. To verify reading compliance and exclude off-target recordings, speech was transcribed using the Google Speech-to-Text API (automatic language detection enabled). Transcripts were aligned with target sentences using a dual-weighted fuzzy matching procedure (FuzzyWuzzy library; 60% token set similarity and 40% edit distance similarity); a composite similarity score above 65% indicated a valid match. All matches were manually reviewed, and ambiguous cases were adjudicated by trained researchers. Transcripts were used solely for quality control and compliance checking; no linguistic features were included in model training. Recordings were then converted from MP3 (44.1 kHz; stereo) to WAV (16 kHz; mono; 16 bits) file formats using FFmpeg. After preprocessing, downstream analyses were conducted exclusively on deidentified acoustic feature values rather than raw speech signals.</p></sec><sec id="s2-5"><title>Acoustic Feature Extraction</title><p>Acoustic features were extracted using openSMILE[<xref ref-type="bibr" rid="ref29">29</xref>] (version 3.0.2; audEERING GmbH [43] with the emobase and emobase2010 configurations, yielding a 570D feature vector for each utterance. The feature space comprised 30 low-level descriptors summarized by 19 functionals, covering intensity and loudness, spectral properties (eg, MFCCs), pitch-related measures (F0 and derived measures), voice quality indexes (jitter, shimmer, and harmonic-to-noise ratio [HNR]), and short-term signal characteristics (eg, zero-crossing rate). Full feature definitions and naming conventions are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-6"><title>Cross-Validation Design and Leakage-Aware Evaluation</title><p>All analyses were implemented in R (version 4.3.3; R Foundation for Statistical Computing). To prevent optimistic bias due to within-participant leakage (ie, multiple recordings per participant), we used participant-level stratified 5-fold cross-validation, assigning all 16 utterances from the same participant to a single fold [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Stratification was performed by case-control status and sex to preserve balance across folds. Age was evaluated as a potential confounder; given the narrow age range and the focus on acoustic-only feature sets, models were primarily trained on speech features.</p><p>Within each training fold, we applied the following preprocessing steps: (1) removal of constant features, (2) elimination of highly correlated feature pairs (|<italic>r</italic>|&#x2265;0.90), and (3) minimum-maximum normalization using parameters derived from the training data and applied unchanged to the corresponding test fold. All data-driven steps&#x2014;including normalization, feature filtering, feature selection, and any hyperparameter tuning&#x2014;were performed using training data only.</p><p>All performance metrics were computed at the participant level (N=96). Participant-level predictions were obtained by averaging the 16 utterance-level feature vectors recorded for each participant to form a single participant-level feature representation, which was then used as input to the model. Accordingly, each confusion matrix adds up to 96.</p></sec><sec id="s2-7"><title>Feature Selection</title><p>Feature selection was conducted within each training fold using a prespecified mixed-ranking strategy integrating 3 criteria: random forest (RF) variable importance (50%), feature-outcome correlation (30%), and ANOVA statistics (20%) [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Features were ranked by aggregated scores, and the top 16 features were retained for model training in each fold. The selected features predominantly captured prosodic, spectral, and voice quality characteristics (eg, MFCC, Line Spectrum Pairs (LSP), F0 and derived measures, and HNR).</p></sec><sec id="s2-8"><title>Model Validation</title><p>We evaluated 7 machine learning algorithms: RF, elastic net generalized linear model (GLMNET), Extreme Gradient Boosting (XGB), support vector machine (SVM; radial basis function kernel), k-nearest neighbor (KNN), naive Bayes (NB), and a feed-forward neural network (NNET).</p><p>Given the modest sample size (N=96), hyperparameter settings were intentionally conservative to reduce overfitting. Analyses were conducted using standard R packages: <italic>randomForest</italic> (RF), <italic>glmnet</italic> (GLMNET), <italic>xgboost</italic> (XGB), <italic>e1071</italic> (SVM), <italic>caret</italic> (KNN and NNET), and <italic>naivebayes</italic> (NB). Hyperparameter settings were as follows: RF used 300 trees with a minimum node size of 10; GLMNET used an &#x03B1; of 0.5 with &#x03BB; selected via inner cross-validation; XGB used max_depth of 3, learning_rate of 0.1, and a subsample of 0.8 with L1 and L2 regularization (&#x03B1;=0.5; &#x03BB;=1); SVM used a radial basis function kernel with C of 1; KNN and NNET were fit via <italic>caret</italic> with limited inner resampling for parameter selection; and NB used default parameters without tuning. All tuning and model fitting was confined to training folds.</p></sec><sec id="s2-9"><title>Performance Evaluation</title><p>Model discrimination was assessed using AUC as the primary metric alongside accuracy, sensitivity, specificity, precision, negative predictive value, and <italic>F</italic><sub>1</sub>-score. Receiver operating characteristic curves and confusion matrices were used to summarize discriminative performance and error patterns. Metrics were defined as follows:</p><disp-formula id="E1"><label>(1)</label><mml:math id="eqn1"><mml:mtext>Accuracy=</mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TP+TN</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP+ TN + FP + FN</mml:mtext></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="E2"><label>(2)</label><mml:math id="eqn2"><mml:mtext>Sensitivity=</mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP+ FN</mml:mtext></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="E3"><label>(3)</label><mml:math id="eqn3"><mml:mtext>Specificity=</mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TN</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TN</mml:mtext><mml:mtext> </mml:mtext><mml:mtext>+ FP</mml:mtext></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="E4"><label>(4)</label><mml:math id="eqn4"><mml:mtext>Precision=</mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext><mml:mtext> </mml:mtext><mml:mtext>+ FP</mml:mtext></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="E5"><label>(5)</label><mml:math id="eqn5"><mml:mtext>NPV =</mml:mtext><mml:mfrac><mml:mrow><mml:mtext>TN</mml:mtext></mml:mrow><mml:mrow><mml:mtext>TN </mml:mtext><mml:mtext> </mml:mtext><mml:mtext>+ FN</mml:mtext></mml:mrow></mml:mfrac></mml:math></disp-formula><disp-formula id="E6"><label>(6)</label><mml:math id="eqn6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mfrac><mml:mrow><mml:mtext>Precision</mml:mtext><mml:mo>&#x00D7;</mml:mo><mml:mtext>Sensitivity</mml:mtext></mml:mrow><mml:mrow><mml:mtext>Precision</mml:mtext><mml:mo>+</mml:mo><mml:mtext>Sensitivity</mml:mtext></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula></sec><sec id="s2-10"><title>Feature Stability</title><p>Feature stability was quantified by the frequency with which each feature was selected across cross-validation folds. Features selected in 80% or more of folds were classified as stable [<xref ref-type="bibr" rid="ref34">34</xref>]. All preprocessing parameters were derived from training data and applied identically to test data to ensure reproducibility.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview</title><p>The final sample comprised 96 participants (n=48, 50% per group) aged 16 to 25 years (mean age 18.76, SD 1.48 years), with 42.7% (n=41) female participants overall. Gender distributions were comparable between groups (&#x03C7;&#x00B2;&#x2081;=0.17; <italic>P</italic>=.68). Age differed between groups, with the SI group being older than the control group (SI: mean 19.25, SD 1.78 years; controls: mean 18.27, SD 0.92 years; <italic>t</italic><sub>94</sub>=&#x2212;3.39; <italic>P</italic>=.001; <xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Demographics of the participants (N=96).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Variables</td><td align="left" valign="bottom">Overall</td><td align="left" valign="bottom">SI<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> group (n=48)</td><td align="left" valign="bottom">Control group (n=48)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD)</td><td align="left" valign="top">18.76 (1.48)</td><td align="left" valign="top">19.25 (1.78)</td><td align="left" valign="top">18.27 (0.92)</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top" colspan="4">Sex, n (%)</td><td align="left" valign="top">.68</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">55 (57.3)</td><td align="left" valign="top">29 (60.4)</td><td align="left" valign="top">26 (54.2)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">41 (42.7)</td><td align="left" valign="top">19 (39.6)</td><td align="left" valign="top">22 (45.8)</td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>SI: suicidal ideation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Model Performance</title><p>As shown in <xref ref-type="table" rid="table2">Table 2</xref>, under participant-level, leakage-aware stratified 5-fold cross-validation, 6 of the 7 models achieved acceptable discrimination (AUCs&#x003E;0.70), whereas NNET performed less well (AUC=0.629). RF yielded the best overall performance (AUC=0.813; accuracy=0.748; sensitivity=0.730; specificity=0.782). NB ranked second (AUC=0.806) and achieved the highest accuracy (0.757). XGB (AUC=0.781) and SVM (AUC=0.772) also showed good discrimination. KNN achieved the highest specificity (0.855) but lower sensitivity (0.645), whereas GLMNET showed more modest discrimination (AUC=0.729). Receiver operating characteristic curves and confusion matrices are shown in <xref ref-type="fig" rid="figure1">Figures 1</xref> and <xref ref-type="fig" rid="figure2">2</xref>, respectively.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Predictive performance of the 7 machine learning models.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Models</td><td align="left" valign="top">AUC<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> (95% CI)</td><td align="left" valign="top">Accuracy</td><td align="left" valign="top">Sensitivity</td><td align="left" valign="top">Specificity</td><td align="left" valign="top">Precision</td><td align="left" valign="top">NPV<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.813 (0.77-0.84)</td><td align="left" valign="top">0.748</td><td align="left" valign="top">0.730</td><td align="left" valign="top">0.782</td><td align="left" valign="top">0.772</td><td align="left" valign="top">0.725</td><td align="left" valign="top">0.747</td></tr><tr><td align="left" valign="top">GLMNET<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.729 (0.64-0.81)</td><td align="left" valign="top">0.671</td><td align="left" valign="top">0.650</td><td align="left" valign="top">0.709</td><td align="left" valign="top">0.697</td><td align="left" valign="top">0.654</td><td align="left" valign="top">0.666</td></tr><tr><td align="left" valign="top">XGB<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">0.781 (0.73-0.83)</td><td align="left" valign="top">0.738</td><td align="left" valign="top">0.770</td><td align="left" valign="top">0.727</td><td align="left" valign="top">0.740</td><td align="left" valign="top">0.752</td><td align="left" valign="top">0.746</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">0.772 (0.71-0.83)</td><td align="left" valign="top">0.748</td><td align="left" valign="top">0.770</td><td align="left" valign="top">0.745</td><td align="left" valign="top">0.763</td><td align="left" valign="top">0.743</td><td align="left" valign="top">0.759</td></tr><tr><td align="left" valign="top">KNN<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">0.757 (0.70-0.80)</td><td align="left" valign="top">0.740</td><td align="left" valign="top">0.645</td><td align="left" valign="top">0.855</td><td align="left" valign="top">0.811</td><td align="left" valign="top">0.691</td><td align="left" valign="top">0.715</td></tr><tr><td align="left" valign="top">NB<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup></td><td align="left" valign="top">0.806 (0.72-0.88)</td><td align="left" valign="top">0.757</td><td align="left" valign="top">0.790</td><td align="left" valign="top">0.745</td><td align="left" valign="top">0.769</td><td align="left" valign="top">0.761</td><td align="left" valign="top">0.770</td></tr><tr><td align="left" valign="top">NNET<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td><td align="left" valign="top">0.629 (0.56-0.72)</td><td align="left" valign="top">0.569</td><td align="left" valign="top">0.430</td><td align="left" valign="top">0.732</td><td align="left" valign="top">0.645</td><td align="left" valign="top">0.548</td><td align="left" valign="top">0.500</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>AUC: area under the curve.</p></fn><fn id="table2fn2"><p><sup>b</sup>NPV: negative predictive value.</p></fn><fn id="table2fn3"><p><sup>c</sup>RF: random forest.</p></fn><fn id="table2fn4"><p><sup>d</sup>GLMNET: elastic net generalized linear model.</p></fn><fn id="table2fn5"><p><sup>e</sup>XGB: Extreme Gradient Boosting.</p></fn><fn id="table2fn6"><p><sup>f</sup>SVM: support vector machine.</p></fn><fn id="table2fn7"><p><sup>g</sup>KNN: k-nearest neighbor.</p></fn><fn id="table2fn8"><p><sup>h</sup>NB: naive Bayes.</p></fn><fn id="table2fn9"><p><sup>i</sup>NNET: feed-forward neural network.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Receiver operating characteristic curves for the 7 machine learning models (random forest [RF], elastic net generalized linear model [GLMNET], Extreme Gradient Boosting [XGB], support vector machine [SVM], k-nearest neighbor [KNN], naive Bayes [NB], and feed-forward neural network [NNET]). RF achieved the highest area under the curve (AUC), whereas NNET showed the lowest AUC.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e92646_fig01.png"/></fig><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Confusion matrices for the 7 models (random forest [RF], elastic net generalized linear model [GLMNET], Extreme Gradient Boosting [XGB], support vector machine [SVM], k-nearest neighbor [KNN], naive Bayes [NB], and feed-forward neural network [NNET]) at the participant level (N=96). Rows indicate true labels (&#x201C;control&#x201D; and &#x201C;suicide&#x201D;), and columns indicate predicted labels (&#x201C;control&#x201D; and &#x201C;suicide&#x201D;); cell values are counts. Treating &#x201C;suicide&#x201D; as the positive class, the 4 cells correspond to true negatives (control correctly identified as control), false positives (control incorrectly identified as suicide), false negatives (suicide cases incorrectly identified as control), and true positives (suicide cases correctly identified as suicide).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e92646_fig02.png"/></fig><p>Beyond aggregate metrics, an analysis of the confusion matrices (<xref ref-type="fig" rid="figure2">Figure 2</xref>) revealed specific patterns of misclassification. For the best-performing model (RF), while it correctly identified 72.9% (35/48) of the SI cases, 27.1% (13/48) of the participants with SI were incorrectly classified as controls (false negatives). Conversely, 20.8% (10/48) of the controls were misclassified as SI (false positives). These results indicate that, while the model captures core acoustic markers of distress, 27.1% (13/48) of SI cases remained undetected, highlighting the current limitations of a purely acoustic approach in this cohort.</p></sec><sec id="s3-3"><title>Feature Stability</title><p>The 20 most frequently selected features across cross-validation folds are shown in Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Eight features were selected in at least 4 folds, indicating relatively consistent selection across resamples. These features predominantly reflected pitch variability (F0-related indexes), spectral shape (MFCC kurtosis), and harmonicity (HNR; Tables S2-S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s3-4"><title>Univariate Associations With SI Status</title><p>To characterize group differences for the robust feature set, we conducted independent-sample 2-tailed <italic>t</italic> tests and correlation analyses with false discovery rate (FDR) correction (Tables S2-S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). After FDR correction, 7 of the 8 robust features remained significantly associated with SI status. Compared with the control group, the SI group showed higher values in 6 features&#x2014;F0_sma_iqr2-3, F0env_sma_stddev, F0env_sma_iqr1-2, F0env_sma_linregerrQ, mfcc_sma [<xref ref-type="bibr" rid="ref5">5</xref>]_kurtosis, and mfcc_sma [<xref ref-type="bibr" rid="ref12">12</xref>]_kurtosis&#x2014;whereas HNR_sma_iqr1-2 was lower in the SI group. Effect sizes were in the medium to large range (Cohen <italic>d</italic>=0.44&#x2010;0.85). Correlation analyses showed a consistent pattern: the 6 F0 variability and MFCC kurtosis features correlated positively with SI status (<italic>r</italic>=0.28&#x2010;0.40), whereas HNR_sma_iqr1-2 was negatively correlated (<italic>r</italic>=&#x2212;0.22). HNR_sma_stddev did not survive FDR correction.</p></sec><sec id="s3-5"><title>Multivariable and Sentence Type Effects</title><p>Logistic regression analyses corroborated the univariate findings. Higher values of F0 variability indexes and MFCC kurtosis were associated with increased odds of SI group membership, whereas HNR_sma_iqr1-2 was associated with decreased odds. To facilitate interpretability, standardized odds ratios (ORs; per 1 SD increase) were reported: MFCC kurtosis features showed standardized ORs of 1.94 and 2.25, standardized ORs for F0-related features ranged from 2.03 to 2.50, and HNR_sma_iqr1-2 showed a standardized OR of 0.62 (Tables S2-S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>Mixed-effects models further indicated that sentence type accounted for substantial variance in most acoustic features (Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Significant group effects were observed for mfcc_sma [<xref ref-type="bibr" rid="ref5">5</xref>]_kurtosis (<italic>F</italic><sub>1,94</sub>=9.048; <italic>P</italic>=.003) and mfcc_sma [<xref ref-type="bibr" rid="ref12">12</xref>]_kurtosis (<italic>F</italic><sub>1,94</sub>=7.718; <italic>P</italic>=.007), with nonsignificant sentence type effects and interactions (<italic>P</italic>&#x003E;.05 in all cases). F0-related features showed both group effects (<italic>F</italic><sub>1,94</sub>=10.296&#x2010;17.373; <italic>P</italic>&#x003C;.001 to <italic>P</italic>=.002) and sentence type effects (<italic>F</italic><sub>3,1434</sub>=13.256&#x2010;75.388; <italic>P</italic>&#x003C;.001) without evidence of group &#x00D7; sentence type interactions. For HNR features, sentence type effects were pronounced (<italic>F</italic><sub>3,1434</sub>=5.684 and 216.477; <italic>P</italic>&#x003C;.001), whereas group effects were weak or marginal (HNR_sma_stddev: <italic>F</italic><sub>1,94</sub>=2.593 and <italic>P</italic>=.11; HNR_sma_iqr1-2: <italic>F</italic><sub>1,94</sub>=4.664 and <italic>P</italic>=.03), and interactions were nonsignificant.</p></sec><sec id="s3-6"><title>Model Interpretability</title><p>To interpret model behavior, we examined RF feature importance and visualized marginal effects using partial dependence plots. F0 variability features ranked among the most influential predictors (Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Partial dependence plots suggested nonlinear relationships between feature values and predicted SI risk for several features (Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), whereas HNR_sma_iqr1-2 showed a nonmonotonic pattern.</p><p>Shapley additive explanations analyses provided convergent evidence regarding the prominence of the robust feature set (Figure S5 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>; <xref ref-type="fig" rid="figure3">Figure 3</xref>). F0_sma_iqr2-3 and F0env_sma_stddev generally contributed positively to SI predictions, whereas HNR-related features tended to contribute negatively. Notably, mfcc_sma [<xref ref-type="bibr" rid="ref12">12</xref>]_kurtosis showed bidirectional Shapley additive explanations values (&#x2212;0.087 to 0.174), and F0env_sma_stddev exhibited the largest spread (&#x2212;0.100 to 0.196), indicating heterogeneity in feature contributions across individuals.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Shapley additive explanations (SHAP) summary plot depicting the effects and directionality of features.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="formative_v10i1e92646_fig03.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study presents a methodologically rigorous proof-of-concept framework for detecting SI from standardized, smartphone-recorded read speech in university students. Speech offers a low-burden complementary behavioral signal that may mitigate reliance on self-disclosure in SI assessment [<xref ref-type="bibr" rid="ref35">35</xref>]. The central contribution is a leakage-resistant modeling and validation pipeline built around (1) standardized prompts, (2) explicit reading compliance checks, (3) participant-level cross-validation to prevent within-speaker contamination, and (4) transparent feature stability and interpretability analyses. These design choices directly address key validity threats in speech-based modeling, particularly speaker leakage and evaluation bias.</p><p>While our results are promising, the interpretation must be matched to our specific methodology and sample characteristics. Notably, the SI group was significantly older than the control group (<italic>P</italic>=.001). Although this age difference was numerically small (approximately 1 year) and occurred within a narrow developmental window (late adolescence), it represents a potential confounding factor that limits the direct attribution of all acoustic variance to psychological status alone. Furthermore, our use of a constrained read speech task, while it improved standardization, may have limited the expression of more intense, spontaneous emotional signals often captured in crisis hotline studies.</p><p>The best-performing model in terms of AUC was RF (AUC=0.813; accuracy=0.748), followed by NB (AUC=0.806; accuracy=0.757). The observed misclassification rate&#x2014;particularly the false negatives&#x2014;merits clinical consideration. Misclassification in SI detection may stem from several factors: (1) the &#x201C;silent&#x201D; or &#x201C;stoic&#x201D; phenotype of ideation where vocal prosody remains intact despite internal distress; (2) individual variability in baseline speaking styles that overlaps with SI-linked features; and (3) the modest sample size (N=96), which may not capture the full heterogeneity of vocal expressions in SI. Consequently, our model should be viewed as a screening aid rather than a definitive diagnostic tool, consistent with the need for cautious deployment of AI in mental health.</p><p>These results fall within the performance range reported in several influential speech-suicide machine learning studies, although differences in population, speech task, labels, and validation design constrain direct numeric comparisons. Su et al [<xref ref-type="bibr" rid="ref22">22</xref>] analyzed crisis hotline calls (naturalistic, emotionally loaded speech with clinician-rated risk labels) and reported the best model performance in terms of accuracy at approximately 0.75, with <italic>F</italic><sub>1</sub>-scores of approximately 0.70. Belouali et al [<xref ref-type="bibr" rid="ref36">36</xref>] used mobile app narrative recordings in US veterans and reported RF performance of approximately 0.80 (AUC) with sensitivity of 86% and specificity of 70% when combining acoustic and linguistic features. Min et al [<xref ref-type="bibr" rid="ref21">21</xref>] evaluated voice from clinical interviews in patients with mood disorders and observed more modest between-person discrimination (AUC&#x2248;0.62) but improved within-person prediction of worsening suicidality over time (accuracy&#x2248;0.79; AUC&#x2248;0.67), highlighting how label definition (cross-sectional discrimination vs longitudinal change) and recording context can shift achievable performance. Pillai et al [<xref ref-type="bibr" rid="ref37">37</xref>] further demonstrated that cross-dataset generalization is frequently poor even when within-dataset metrics appear strong, indicating that distribution shift must be considered alongside algorithm choice.</p><p>Compared with the standardized reading tasks widely used in suicide prediction research based on acoustic features, such as the rainbow passage [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>], our standardized reading materials incorporated a wider variety of sentence structures and sentence functions. The sentences were relatively more independent from one another and exhibited greater variability across items. Furthermore, compared with the reading passages commonly used in English-speaking contexts, our materials were better suited to Chinese phonological characteristics, thereby providing a standardized reading task for populations with Chinese linguistic backgrounds.</p><p>In this context, the observation that a brief read speech task&#x2014;more constrained than narratives or crisis calls&#x2014;still yielded discrimination comparable to that of several naturalistic speech approaches suggests that standardization can retain clinically relevant acoustic variability while improving controllability and reproducibility [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>A major methodological strength of the present work is the explicit handling of within-participant leakage introduced by multiple recordings per speaker (16 utterances per participant). Leakage is a known failure mode in speech machine learning: if folds are split by recordings rather than participants, models can inadvertently learn speaker identity or stable vocal traits, inflating apparent performance [<xref ref-type="bibr" rid="ref40">40</xref>]. Participant-level stratified cross-validation&#x2014;where all utterances from a participant remain in one fold&#x2014;directly addressed this issue and is consistent with best practice recommendations in behavioral machine learning and speech-based mental health research. In addition, the compliance pipeline (automatic transcription plus fuzzy alignment and manual review) reduced noise from off-prompt or incomplete readings, which can otherwise obscure acoustic signal in smartphone audio collected outside laboratory conditions. Such quality control procedures are often underdescribed in the literature but are likely to become increasingly important as speech data collection shifts toward real-world mobile settings [<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>The feature patterns most consistently associated with SI&#x2014;indexes of pitch variability (F0-related measures), spectral shape (MFCC kurtosis), and harmonicity (HNR)&#x2014;are broadly consistent with prior work linking suicidality and related affective states to altered prosody and voice quality while also illustrating how different speech tasks may privilege different acoustic cues. In hotline settings, spontaneous emotional speech likely amplifies prosodic instability and arousal-linked voice changes, whereas standardized reading reduces linguistic variability and may highlight subtler motor or phonatory correlates [<xref ref-type="bibr" rid="ref22">22</xref>]. In narrative recordings, acoustic and linguistic features jointly contribute to psychiatric classification, suggesting that, when free speech is available, content-related information can be additive; standardized prompts isolate acoustics, making inferred associations more interpretable as vocal correlates rather than mixed vocal-semantic signals [<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. Reviews of speech processing in depression and suicidality similarly emphasize recurrent involvement of pitch, energy, spectral measures (often MFCC derived), and voice quality features while noting that effect direction and magnitude can vary by task, language, and recording conditions [<xref ref-type="bibr" rid="ref15">15</xref>].</p><p>The comparability of performance estimates across studies depends not only on metrics but also on sampling strategies and outcome construction. Hotline and telehealth studies often involve higher base rates of distress and labels reflecting clinician-rated risk strata rather than strictly defined ideation thresholds, which can influence separability and reported metrics [<xref ref-type="bibr" rid="ref22">22</xref>]. Conversely, balanced case-control designs improve statistical efficiency for early-stage method development but depart from community prevalence distributions. Consequently, accuracy in a balanced sample should not be interpreted as an expected positive predictive value for population screening without recalibration [<xref ref-type="bibr" rid="ref43">43</xref>]. These considerations underscore the importance of transparent reporting and leakage-aware validation, and they help explain why methodological heterogeneity in recording contexts and analytic pipelines continues to challenge transportability across settings [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>Methodologically, the pattern observed in this study&#x2014;strong performance of simpler, well-regularized models (RF and NB) alongside weaker performance of the neural network&#x2014;aligns with prior observations that more complex architectures can overfit in small, heterogeneous datasets without extensive harmonization [<xref ref-type="bibr" rid="ref21">21</xref>]. Consistent with this, cross-dataset evidence indicates that robustness under distribution shift remains a central bottleneck even when within-dataset metrics appear strong, highlighting the need for evaluation designs and dataset diversity that explicitly test generalizability [<xref ref-type="bibr" rid="ref37">37</xref>].</p><p>In summary, the present work contributes a reproducible methodological template for SI detection from voice by combining standardized prompts, compliance quality control, leakage-aware participant-level validation, and transparent feature stability analyses. When viewed alongside hotline, clinical interview, and narrative speech studies, these results support the feasibility of extracting suicidality-relevant information from speech across diverse contexts while underscoring the importance of rigorous evaluation and cross-context testing as the next step for the field.</p></sec><sec id="s4-2"><title>Limitations</title><p>Several limitations merit consideration. First, the cross-sectional nature and specific recruitment from a single university limit the generalizability of our findings. The identified acoustic markers should be interpreted as correlational signals rather than markers of future suicide risk. As highlighted by the issue of misclassification in our sample, these features alone are insufficient for high-stakes clinical decision-making without integration into a broader, multimodal assessment framework. Second, while the standardized read speech task controlled linguistic content and reduced between-speaker variability, it may have constrained the range of emotional and expressive speech characteristics captured. Accordingly, the ecological validity of the identified acoustic patterns may be lower than that achievable with spontaneous or emotion-elicited speech. Third, the neural network&#x2013;based model showed weaker performance than the simpler models in this dataset, plausibly reflecting the moderate sample size and the challenges that complex models face in learning stable, generalizable representations from heterogeneous speech data. Fourth, our pipeline relies exclusively on acoustic features derived from standardized read speech, ignoring linguistic content and physiological biomarkers. These limitations are consistent with concerns raised in recent reviews and empirical work on acoustic-based suicide risk assessment [<xref ref-type="bibr" rid="ref20">20</xref>].</p></sec><sec id="s4-3"><title>Future Directions</title><p>Future research should strengthen the construct validity of acoustic indicators and evaluate their sensitivity to individual differences in SI. First, expanding recruitment to multiple universities and larger samples will improve statistical power and enable testing of whether the identified acoustic features consistently reflect underlying psychological constructs&#x2014;such as emotional dysregulation, cognitive constriction, or motivational withdrawal&#x2014;across diverse student populations. Second, future studies should explicitly model individual differences in speech production and psychological functioning. Beyond group-level classification, person-centered and longitudinal designs may help disentangle stable, trait-like vocal characteristics from state-dependent fluctuations associated with changes in suicidal risk, thereby improving interpretability and reducing misclassification driven by normative speaking style variability. Third, efforts to develop early warning systems should remain grounded in psychological theory and measurement principles. Acoustic features are best evaluated as indicators within a multi-method assessment framework, complementing self-report and clinical judgment rather than serving as stand-alone predictors. Aligning model outputs with interpretable psychological constructs may facilitate responsible integration into prevention contexts while preserving conceptual clarity and ethical safeguards. Fourth, moving beyond the unimodal acoustic approach, future research should explore multimodal fusion that integrates speech acoustics with natural language processing and physiological biomarkers. Multimodal fusion may be a superior method for detecting SI as it captures complementary risk dimensions&#x2014;such as lexical markers of hopelessness, prosodic instability, and autonomic arousal&#x2014;while compensating for the limitations of any single modality.</p></sec><sec id="s4-4"><title>Conclusions</title><p>This study demonstrates the feasibility of using smartphone-acquired acoustic features from standardized read speech to discriminate SI among Chinese university students. The RF model showed promising preliminary performance (AUC=0.813) and identified stable acoustic markers. These findings support speech-derived acoustics as a low-burden, objective complement to traditional self-report screening in campus mental health contexts, where concealment and resource constraints can limit detection. Importantly, acoustic indicators should be interpreted as correlational signals rather than diagnostic tools, and their use should be embedded within existing counseling pathways and ethical safeguards. With further external validation and evaluation in more diverse, real-world settings, speech-based acoustic analysis can potentially alleviate the burden on campus counseling centers by providing an automated, preliminary triage stage in a multimodal screening framework.</p></sec></sec></body><back><ack><p>The authors sincerely thank all participating students for their time and contributions to this study. Their willingness to complete the questionnaires and provide speech recordings made this research possible. The authors declare the use of generative AI (GenAI) in the research and writing process. According to the Generative AI Delegation Taxonomy (2025), the following tasks were delegated to GenAI tools under full human supervision: proofreading and editing. The GenAI tool used was Google Gemini 3. Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This study was supported by the Chongqing Science and Technology Bureau as part of the 2023 Public Safety Technology Innovation and Application Development Special Project (project title: "Social Mentality Intelligent Ecological Monitoring and Intervention Research"; grant CSTB2023TIAD-KPX0064; total project budget: CNY 1,000,000 [CNY 1=US $0.15 as of August 1, 2026]).</p></sec><sec><title>Data Availability</title><p>The data underpinning the findings of this study encompass sensitive personal information. Consequently, they are not publicly accessible owing to privacy and ethical constraints. Deidentified data may be provided by the corresponding author upon a reasonable request and subject to appropriate ethics approval.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: ML, ZF</p><p>Data curation: ML, LT, FL, JQ, TH, ZZ, TJ, ZL</p><p>Formal analysis: ML, LT, JL</p><p>Funding acquisition: ML, ZF</p><p>Investigation: ML, FL, JQ, TH, TJ, ZL</p><p>Methodology: ML, LT, JL, ZZ</p><p>Project administration: ML, ZF, JZ</p><p>Resources: ML, ZF, JZ, JX, HH</p><p>Supervision: ML, ZF, JZ</p><p>Validation: ML, LT, JL</p><p>Visualization: ML, LT</p><p>Writing&#x2014;original draft: ML, LT</p><p>Writing&#x2014;review and editing: ML, LT, JX, HH, FL, JQ, TH, JL, ZZ, TJ, ZL, XW, JZ, ZF</p><p>ZF serves as the guarantor of this study and takes responsibility for the integrity of the work as a whole, from study design and data collection to analysis and reporting.</p><p>ZF is the corresponding author. JZ is the co-corresponding author.</p><p>ML, LT, JX, and HH are co-first authors.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb2">F0</term><def><p> fundamental frequency</p></def></def-item><def-item><term id="abb3">FDR</term><def><p>false discovery rate</p></def></def-item><def-item><term id="abb4">GLMNET</term><def><p>elastic net generalized linear model</p></def></def-item><def-item><term id="abb5">HNR </term><def><p>harmonic-to-noise ratio</p></def></def-item><def-item><term id="abb6">KNN</term><def><p> k-nearest neighbor</p></def></def-item><def-item><term id="abb7">LSP</term><def><p>Line Spectrum Pairs</p></def></def-item><def-item><term id="abb8">MFCC </term><def><p>mel-frequency cepstral coefficient</p></def></def-item><def-item><term id="abb9">NB</term><def><p>naive Bayes</p></def></def-item><def-item><term id="abb10">NNET </term><def><p>feed-forward neural network</p></def></def-item><def-item><term id="abb11">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb12">RF </term><def><p>random forest</p></def></def-item><def-item><term id="abb13">SI </term><def><p>suicidal ideation</p></def></def-item><def-item><term id="abb14">SIOSS</term><def><p>Self-Rating Idea of Suicide Scale</p></def></def-item><def-item><term id="abb15">SVM</term><def><p> support vector machine</p></def></def-item><def-item><term id="abb16">XGB </term><def><p>Extreme Gradient Boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="web"><article-title>Suicide</article-title><source>World Health Organization</source><year>2025</year><access-date>2025-11-15</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/news-room/fact-sheets/detail/suicide">https://www.who.int/news-room/fact-sheets/detail/suicide</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lamontagne</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Zabala</surname><given-names>PK</given-names> </name><name name-style="western"><surname>Zarate</surname><given-names>CA Jr</given-names> </name><name name-style="western"><surname>Ballard</surname><given-names>ED</given-names> </name></person-group><article-title>Toward objective characterizations of suicide risk: a narrative review of laboratory-based cognitive and behavioral tasks</article-title><source>Neurosci Biobehav Rev</source><year>2023</year><month>10</month><volume>153</volume><fpage>105361</fpage><pub-id pub-id-type="doi">10.1016/j.neubiorev.2023.105361</pub-id><pub-id pub-id-type="medline">37595649</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sommers-Flanagan</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shaw</surname><given-names>SL</given-names> </name></person-group><article-title>Suicide risk assessment: what psychologists should know</article-title><source>Prof Psychol Res Pract</source><year>2017</year><volume>48</volume><issue>2</issue><fpage>98</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.1037/pro0000106</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hamdan-Mansour</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Hamdan-Mansour</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Allaham</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Alrashidi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alhaiti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mansour</surname><given-names>LA</given-names> </name></person-group><article-title>Academic procrastination, loneliness, and academic anxiety as predictors of suicidality among university students</article-title><source>Int J Ment Health Nurs</source><year>2024</year><month>12</month><volume>33</volume><issue>6</issue><fpage>2054</fpage><lpage>2062</lpage><pub-id pub-id-type="doi">10.1111/inm.13366</pub-id><pub-id pub-id-type="medline">38797963</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akram</surname><given-names>U</given-names> </name><name name-style="western"><surname>Ypsilanti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gardani</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Prevalence and psychiatric correlates of suicidal ideation in UK university students</article-title><source>J Affect Disord</source><year>2020</year><month>07</month><day>1</day><volume>272</volume><fpage>191</fpage><lpage>197</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2020.03.185</pub-id><pub-id pub-id-type="medline">32379615</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akram</surname><given-names>U</given-names> </name><name name-style="western"><surname>Irvine</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gardani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Allen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Akram</surname><given-names>A</given-names> </name><name name-style="western"><surname>Stevenson</surname><given-names>JC</given-names> </name></person-group><article-title>Prevalence of anxiety, depression, mania, insomnia, stress, suicidal ideation, psychotic experiences, &#x0026; loneliness in UK university students</article-title><source>Sci Data</source><year>2023</year><month>09</month><day>13</day><volume>10</volume><issue>1</issue><fpage>621</fpage><pub-id pub-id-type="doi">10.1038/s41597-023-02520-5</pub-id><pub-id pub-id-type="medline">37704598</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akram</surname><given-names>U</given-names> </name><name name-style="western"><surname>Drabble</surname><given-names>J</given-names> </name><name name-style="western"><surname>Irvine</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Prevalence and psychiatric correlates of loneliness in UK university students</article-title><source>Npj Ment Health Res</source><year>2025</year><month>09</month><day>16</day><volume>4</volume><issue>1</issue><fpage>45</fpage><pub-id pub-id-type="doi">10.1038/s44184-025-00144-8</pub-id><pub-id pub-id-type="medline">40957876</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Loneliness, internalizing and externalizing problems, and suicidal ideation among Chinese adolescents: a longitudinal mediation analysis</article-title><source>J Adolesc Health</source><year>2025</year><month>01</month><volume>76</volume><issue>1</issue><fpage>96</fpage><lpage>104</lpage><pub-id pub-id-type="doi">10.1016/j.jadohealth.2024.08.010</pub-id><pub-id pub-id-type="medline">39365230</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>T</given-names> </name><name name-style="western"><surname>He</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>L</given-names> </name><etal/></person-group><article-title>The relationships between anxiety and suicidal ideation and between depression and suicidal ideation among Chinese college students: a network analysis</article-title><source>Heliyon</source><year>2023</year><month>10</month><volume>9</volume><issue>10</issue><fpage>e20938</fpage><pub-id pub-id-type="doi">10.1016/j.heliyon.2023.e20938</pub-id><pub-id pub-id-type="medline">37876446</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bornheimer</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Czyz</surname><given-names>E</given-names> </name><name name-style="western"><surname>Koo</surname><given-names>HJ</given-names> </name><etal/></person-group><article-title>Suicide risk profiles and barriers to professional help-seeking among college students with elevated risk for suicide</article-title><source>J Psychiatr Res</source><year>2022</year><month>08</month><volume>152</volume><fpage>305</fpage><lpage>312</lpage><pub-id pub-id-type="doi">10.1016/j.jpsychires.2022.06.028</pub-id><pub-id pub-id-type="medline">35772258</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Social support and suicide risk among Chinese university students: a mental health perspective</article-title><source>Front Public Health</source><year>2021</year><volume>9</volume><fpage>566993</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2021.566993</pub-id><pub-id pub-id-type="medline">33681117</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>M</given-names> </name><name name-style="western"><surname>Xin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Risk factors for suicidality among college students: a systematic review and meta-analysis</article-title><source>J Affect Disord</source><year>2025</year><month>08</month><day>1</day><volume>382</volume><fpage>567</fpage><lpage>578</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2025.04.137</pub-id><pub-id pub-id-type="medline">40280440</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Newman</surname><given-names>G</given-names> </name></person-group><article-title>Suicide risks among U.S. college students: a time-series cross-sectional study examining institutional characteristics and behavioral factors</article-title><source>Prev Sci</source><year>2025</year><month>12</month><volume>26</volume><issue>8</issue><fpage>1169</fpage><lpage>1182</lpage><pub-id pub-id-type="doi">10.1007/s11121-025-01854-3</pub-id><pub-id pub-id-type="medline">41266904</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>L&#x00E1;zaro-P&#x00E9;rez</surname><given-names>C</given-names> </name><name name-style="western"><surname>Munuera G&#x00F3;mez</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mart&#x00ED;nez-L&#x00F3;pez</surname><given-names>J&#x00C1;</given-names> </name><name name-style="western"><surname>G&#x00F3;mez-Gal&#x00E1;n</surname><given-names>J</given-names> </name></person-group><article-title>Predictive factors of suicidal ideation in Spanish university students: a health, preventive, social, and cultural approach</article-title><source>J Clin Med</source><year>2023</year><month>02</month><day>2</day><volume>12</volume><issue>3</issue><fpage>1207</fpage><pub-id pub-id-type="doi">10.3390/jcm12031207</pub-id><pub-id pub-id-type="medline">36769853</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cummins</surname><given-names>N</given-names> </name><name name-style="western"><surname>Scherer</surname><given-names>S</given-names> </name><name name-style="western"><surname>Krajewski</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schnieder</surname><given-names>S</given-names> </name><name name-style="western"><surname>Epps</surname><given-names>J</given-names> </name><name name-style="western"><surname>Quatieri</surname><given-names>TF</given-names> </name></person-group><article-title>A review of depression and suicide risk assessment using speech analysis</article-title><source>Speech Commun</source><year>2015</year><month>07</month><volume>71</volume><fpage>10</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1016/j.specom.2015.03.004</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Figueroa</surname><given-names>C</given-names> </name><name name-style="western"><surname>Guill&#x00E9;n</surname><given-names>V</given-names> </name><name name-style="western"><surname>Huenup&#x00E1;n</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Comparison of acoustic parameters of voice and speech according to vowel type and suicidal risk in adolescents</article-title><source>J Voice</source><year>2024</year><month>08</month><day>30</day><fpage>S0892-1997(24)00254-6</fpage><pub-id pub-id-type="doi">10.1016/j.jvoice.2024.08.006</pub-id><pub-id pub-id-type="medline">39217086</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kappen</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vanhollebeke</surname><given-names>G</given-names> </name><name name-style="western"><surname>Van Der Donckt</surname><given-names>J</given-names> </name><name name-style="western"><surname>Van Hoecke</surname><given-names>S</given-names> </name><name name-style="western"><surname>Vanderhasselt</surname><given-names>MA</given-names> </name></person-group><article-title>Acoustic and prosodic speech features reflect physiological stress but not isolated negative affect: a multi-paradigm study on psychosocial stressors</article-title><source>Sci Rep</source><year>2024</year><month>03</month><day>6</day><volume>14</volume><issue>1</issue><fpage>5515</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-55550-3</pub-id><pub-id pub-id-type="medline">38448417</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>J</given-names> </name></person-group><article-title>A machine learning-based case-control study on suicide risk identification: integrating acoustic and linguistic features under stress conditions</article-title><source>Depress Anxiety</source><year>2025</year><volume>2025</volume><issue>1</issue><fpage>1671972</fpage><pub-id pub-id-type="doi">10.1155/da/1671972</pub-id><pub-id pub-id-type="medline">40821764</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>MacPherson</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Abur</surname><given-names>D</given-names> </name><name name-style="western"><surname>Stepp</surname><given-names>CE</given-names> </name></person-group><article-title>Acoustic measures of voice and physiologic measures of autonomic arousal during speech as a function of cognitive load</article-title><source>J Voice</source><year>2017</year><month>07</month><volume>31</volume><issue>4</issue><fpage>504.e1</fpage><pub-id pub-id-type="doi">10.1016/j.jvoice.2016.10.021</pub-id><pub-id pub-id-type="medline">27939119</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marie</surname><given-names>A</given-names> </name><name name-style="western"><surname>Garnier</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bertin</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Acoustic and machine learning methods for speech-based suicide risk assessment: a systematic review</article-title><source>J Affect Disord</source><year>2026</year><month>02</month><day>1</day><volume>394</volume><issue>Pt B</issue><fpage>120569</fpage><pub-id pub-id-type="doi">10.1016/j.jad.2025.120569</pub-id><pub-id pub-id-type="medline">41203082</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Min</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>D</given-names> </name><name name-style="western"><surname>Rhee</surname><given-names>SJ</given-names> </name><etal/></person-group><article-title>Acoustic analysis of speech for screening for suicide risk: machine learning classifiers for between- and within-person evaluation of suicidality</article-title><source>J Med Internet Res</source><year>2023</year><month>03</month><day>23</day><volume>25</volume><fpage>e45456</fpage><pub-id pub-id-type="doi">10.2196/45456</pub-id><pub-id pub-id-type="medline">36951913</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Su</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hou</surname><given-names>X</given-names> </name><name name-style="western"><surname>Su</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>L</given-names> </name></person-group><article-title>Acoustic features for identifying suicide risk in crisis hotline callers: machine learning approach</article-title><source>J Med Internet Res</source><year>2025</year><month>04</month><day>14</day><volume>27</volume><fpage>e67772</fpage><pub-id pub-id-type="doi">10.2196/67772</pub-id><pub-id pub-id-type="medline">40228243</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krautz</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Volkening</surname><given-names>J</given-names> </name><name name-style="western"><surname>Raue</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Prediction of suicide using web based voice recordings analyzed by artificial intelligence</article-title><source>Sci Rep</source><year>2025</year><month>07</month><day>4</day><volume>15</volume><issue>1</issue><fpage>23855</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-08639-2</pub-id><pub-id pub-id-type="medline">40615574</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ding</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>Speech based suicide risk recognition for crisis intervention hotlines using explainable multi-task learning</article-title><source>J Affect Disord</source><year>2025</year><month>02</month><day>1</day><volume>370</volume><fpage>392</fpage><lpage>400</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2024.11.022</pub-id><pub-id pub-id-type="medline">39528146</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Speech feature identification model for depressed individuals with suicidal ideation based on autobiographical memory</article-title><source>BMC Psychiatry</source><year>2025</year><month>11</month><day>24</day><volume>25</volume><issue>1</issue><fpage>1154</fpage><pub-id pub-id-type="doi">10.1186/s12888-025-07635-0</pub-id><pub-id pub-id-type="medline">41286790</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Stasak</surname><given-names>B</given-names> </name><name name-style="western"><surname>Epps</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schatten</surname><given-names>HT</given-names> </name><name name-style="western"><surname>Miller</surname><given-names>IW</given-names> </name><name name-style="western"><surname>Provost</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Armey</surname><given-names>MF</given-names> </name></person-group><article-title>Read speech voice quality and disfluency in individuals with recent suicidal ideation or suicide attempt</article-title><source>Speech Commun</source><year>2021</year><month>09</month><volume>132</volume><fpage>10</fpage><lpage>20</lpage><pub-id pub-id-type="doi">10.1016/j.specom.2021.05.004</pub-id><pub-id pub-id-type="medline">42079368</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Torous</surname><given-names>J</given-names> </name><name name-style="western"><surname>Andersson</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bertagnoli</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Towards a consensus around standards for smartphone apps and digital mental health</article-title><source>World Psychiatry</source><year>2019</year><month>02</month><volume>18</volume><issue>1</issue><fpage>97</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1002/wps.20592</pub-id><pub-id pub-id-type="medline">30600619</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="web"><article-title>Ethics and governance of artificial intelligence for health: guidance on large multi-modal models</article-title><source>World Health Organization</source><year>2025</year><access-date>2026-08-03</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/9789240084759">https://www.who.int/publications/i/item/9789240084759</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Eyben</surname><given-names>F</given-names> </name><name name-style="western"><surname>W&#x00F6;llmer</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schuller</surname><given-names>B</given-names> </name></person-group><article-title>openSMILE: the Munich versatile and fast open-source audio feature extractor</article-title><source>MM &#x2019;10: Proceedings of the 18th ACM International Conference on Multimedia</source><year>2010</year><publisher-name>Association for Computing Machinery</publisher-name><pub-id pub-id-type="doi">10.1145/1873951.1874246</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Titze</surname><given-names>IR</given-names> </name></person-group><article-title>Physiologic and acoustic differences between male and female voices</article-title><source>J Acoust Soc Am</source><year>1989</year><month>04</month><volume>85</volume><issue>4</issue><fpage>1699</fpage><lpage>1707</lpage><pub-id pub-id-type="doi">10.1121/1.397959</pub-id><pub-id pub-id-type="medline">2708686</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Y&#x00FC;cesoy</surname><given-names>E</given-names> </name></person-group><article-title>Gender recognition based on the stacking of different acoustic features</article-title><source>Appl Sci</source><year>2024</year><volume>14</volume><issue>15</issue><fpage>6564</fpage><pub-id pub-id-type="doi">10.3390/app14156564</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Albaldawi</surname><given-names>WS</given-names> </name><name name-style="western"><surname>Almuttairi</surname><given-names>RM</given-names> </name></person-group><article-title>Hybrid ANOVA and LASSO methods for feature selection and linear support vector, multilayer perceptron and random forest classifiers based on Spark environment for microarray data classification</article-title><source>IOP Conf Ser Mater Sci Eng</source><year>2021</year><month>02</month><day>1</day><volume>1094</volume><issue>1</issue><fpage>012107</fpage><pub-id pub-id-type="doi">10.1088/1757-899X/1094/1/012107</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name></person-group><article-title>A high-value mobile communication user prediction method based on effective feature selection</article-title><source>J Wuhan Univ Sci Technol</source><year>2017</year><volume>40</volume><issue>2</issue><fpage>149</fpage><pub-id pub-id-type="doi">10.3969/j.issn.1674-3644.2017.02.013</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nogueira</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sechidis</surname><given-names>K</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>G</given-names> </name></person-group><article-title>On the stability of feature selection algorithms</article-title><source>J Mach Learn Res</source><year>2018</year><access-date>2026-08-03</access-date><volume>18</volume><issue>174</issue><fpage>1</fpage><lpage>54</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/v18/17-514.html">https://jmlr.org/papers/v18/17-514.html</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Obegi</surname><given-names>JH</given-names> </name></person-group><article-title>How common is recent denial of suicidal ideation among ideators, attempters, and suicide decedents? A literature review</article-title><source>Gen Hosp Psychiatry</source><year>2021</year><volume>72</volume><fpage>92</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1016/j.genhosppsych.2021.07.009</pub-id><pub-id pub-id-type="medline">34358807</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Belouali</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gupta</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sourirajan</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Acoustic and language analysis of speech for suicidal ideation among US veterans</article-title><source>BioData Min</source><year>2021</year><month>02</month><day>2</day><volume>14</volume><issue>1</issue><fpage>11</fpage><pub-id pub-id-type="doi">10.1186/s13040-021-00245-y</pub-id><pub-id pub-id-type="medline">33531048</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pillai</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nepal</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Investigating generalizability of speech-based suicidal ideation detection using mobile phones</article-title><source>Proc ACM Interact Mob Wearable Ubiquitous Technol</source><year>2023</year><month>12</month><day>19</day><volume>7</volume><issue>4</issue><fpage>1</fpage><lpage>38</lpage><pub-id pub-id-type="doi">10.1145/3631452</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wahidah</surname><given-names>NN</given-names> </name><name name-style="western"><surname>Wilkes</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Salomon</surname><given-names>RM</given-names> </name></person-group><article-title>Timing patterns of speech as potential indicators of near-term suicidal risk</article-title><source>Int J Multidiscip Curr Res</source><year>2015</year><access-date>2026-08-11</access-date><volume>3</volume><issue>6</issue><fpage>1104</fpage><lpage>1114</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://ijmcr.com/index.php/ijmcr/article/view/03.06.02">https://ijmcr.com/index.php/ijmcr/article/view/03.06.02</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Yingthawornsuk</surname><given-names>T</given-names> </name><name name-style="western"><surname>Shiavi</surname><given-names>RG</given-names> </name></person-group><article-title>Distinguishing depression and suicidal risk in men using GMM based frequency contents of affective vocal tract response</article-title><source>2008 International Conference on Control, Automation and Systems</source><year>2008</year><publisher-name>IEEE</publisher-name><pub-id pub-id-type="doi">10.1109/ICCAS.2008.4694621</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Low</surname><given-names>DM</given-names> </name><name name-style="western"><surname>Bentley</surname><given-names>KH</given-names> </name><name name-style="western"><surname>Ghosh</surname><given-names>SS</given-names> </name></person-group><article-title>Automated assessment of psychiatric disorders using speech: a systematic review</article-title><source>Laryngoscope Investig Otolaryngol</source><year>2020</year><month>01</month><volume>5</volume><issue>1</issue><fpage>96</fpage><lpage>116</lpage><pub-id pub-id-type="doi">10.1002/lio2.354</pub-id><pub-id pub-id-type="medline">32128436</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dobbs</surname><given-names>MF</given-names> </name><name name-style="western"><surname>McGowan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Selloni</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Linguistic correlates of suicidal ideation in youth at clinical high-risk for psychosis</article-title><source>Schizophr Res</source><year>2023</year><month>09</month><volume>259</volume><fpage>20</fpage><lpage>27</lpage><pub-id pub-id-type="doi">10.1016/j.schres.2023.03.014</pub-id><pub-id pub-id-type="medline">36933977</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Crocamo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Cioni</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Canestro</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Acoustic and natural language markers for bipolar disorder: a pilot, mHealth cross-sectional study</article-title><source>JMIR Form Res</source><year>2025</year><month>04</month><day>16</day><volume>9</volume><fpage>e65555</fpage><pub-id pub-id-type="doi">10.2196/65555</pub-id><pub-id pub-id-type="medline">40239203</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Posner</surname><given-names>K</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>GK</given-names> </name><name name-style="western"><surname>Stanley</surname><given-names>B</given-names> </name><etal/></person-group><article-title>The Columbia-Suicide Severity Rating Scale: initial validity and internal consistency findings from three multisite studies with adolescents and adults</article-title><source>Am J Psychiatry</source><year>2011</year><month>12</month><volume>168</volume><issue>12</issue><fpage>1266</fpage><lpage>1277</lpage><pub-id pub-id-type="doi">10.1176/appi.ajp.2011.10111704</pub-id><pub-id pub-id-type="medline">22193671</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary materials including 5 figures, 4 tables, and 1 Excel file (list of acoustic feature names).</p><media xlink:href="formative_v10i1e92646_app1.docx" xlink:title="DOCX File, 809 KB"/></supplementary-material></app-group></back></article>