Full text
51,844 characters
· extracted from
preprint-html
· click to expand
AI-Powered Triage of Suicidal Ideation in Adolescents: A Comparative Evaluation of Large Language Models Using Synthetic Clinical Vignettes | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search AI-Powered Triage of Suicidal Ideation in Adolescents: A Comparative Evaluation of Large Language Models Using Synthetic Clinical Vignettes View ORCID Profile Masab Mansoor doi: https://doi.org/10.1101/2025.08.05.25333046 Masab Mansoor 1 Edward Via College of Osteopathic Medicine--Louisiana Campus , Monroe, Louisiana Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Masab Mansoor For correspondence: mmansoor{at}vcom.edu Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Objective To evaluate the performance of leading Large Language Models (LLMs) in classifying suicide risk and generating clinically appropriate action plans for adolescent psychiatric cases presented through synthetic clinical vignettes. Methods We developed 40 synthetic clinical vignettes depicting adolescents with varying levels of suicide risk, structured according to established clinical formulation principles. A gold standard for risk level, based on the Columbia-Suicide Severity Rating Scale (C-SSRS) framework, and corresponding clinical actions was established for each vignette by a panel of two board-certified child and adolescent psychiatrists. Three LLMs (GPT-4o, Claude 3.5 Sonnet, Llama-3.1-70B) were prompted using a structured chain-of-thought methodology to classify risk and propose a detailed action plan. Performance was assessed using quantitative classification metrics (accuracy, precision, recall, F1-score) and qualitative thematic analysis of the generated action plans. Results Quantitative analysis of risk classification revealed variable performance. GPT-4o achieved the highest accuracy (82.5%), followed by Claude 3.5 Sonnet (75.0%) and Llama-3.1-70B (67.5%). F1-scores demonstrated challenges in correctly identifying higher-risk categories, particularly for nuanced presentations of intent. Qualitative thematic analysis of the action plans identified consistent adherence to basic safety protocols (e.g., recommending emergency evaluation for high-risk cases). However, significant and critical failures were pervasive, including the omission of crucial inquiries about access to lethal means, failure to incorporate protective factors into planning, and the generation of clinically inappropriate therapeutic reassurance in a triage context. Conclusions While LLMs demonstrate a nascent ability to process clinical information for suicide risk assessment, significant deficits in clinical reasoning and safety planning persist. Their performance on idealized synthetic data suggests these models are not yet suitable for autonomous clinical decision-making. These findings underscore the imperative for rigorous, clinically-grounded evaluation frameworks and the development of human-in-the-loop systems to ensure patient safety in any future deployment. What is already known on this topic Suicide is a leading cause of death in adolescents, yet current clinical risk assessment tools and subjective judgments have limited predictive accuracy and are difficult to scale in the face of rising demand and workforce shortages. What this study adds This study provides a direct comparative evaluation of multiple state-of-the-art Large Language Models on a standardized adolescent suicide risk triage task, using a synthetic data methodology that allows for controlled assessment of both classification accuracy and the clinical appropriateness of generated action plans. How this study might affect research, practice or policy The findings highlight the potential of LLMs as adjunctive tools in non-specialist settings but also reveal critical safety and reliability gaps that must be addressed through further research, the development of ethical guidelines, and regulatory oversight before any clinical implementation can be considered. Introduction The Escalating Crisis of Adolescent Suicide Suicide represents a profound and escalating public health crisis, particularly among young people. Globally, it is the third leading cause of death for adolescents aged 15–29. 1 In the United States, the statistics are similarly dire, with suicide ranking as the second leading cause of death for individuals aged 10–34. 2 Data from the Centers for Disease Control and Prevention (CDC) indicate that suicide deaths among 10- to 24-year-olds increased by 62% from 2007 to 2021. 4 This trend is paralleled by a rise in suicidal ideation and attempts; in 2023, one in five high school students seriously considered attempting suicide 4 , and 9% made at least one attempt in the past year. 2 This crisis does not affect all populations equally. Disparities are stark, particularly for LGBTQ+ youth, who report significantly higher rates of suicidal ideation (39%) and attempts (12%) compared to their cisgender and heterosexual peers. 5 Youth of color, including Native/Indigenous (24% attempted) and Black/African American (14% attempted) youth, also face disproportionately high rates. 5 This surge in suicidality occurs alongside a broader decline in adolescent mental health, with nearly 1 in 5 teens experiencing a major depressive episode in the past year. 6 The connection is clear: untreated or undertreated mental illness is a primary driver of suicide risk. 1 Deficiencies in Current Risk Assessment Paradigms Despite the urgency of this crisis, the “standard of care” for suicide risk assessment is fraught with limitations. The process heavily relies on a clinician’s subjective judgment, a method with notoriously poor predictive accuracy for a low-base-rate event like suicide. 7 It is widely acknowledged that predicting whether a specific individual will die by suicide is impossible; the clinical goal is assessment of risk to guide intervention, not a definitive forecast. 7 Validated screening instruments, such as the Ask Suicide-Screening Questions (ASQ) and the Columbia-Suicide Severity Rating Scale (C-SSRS), have been widely adopted to standardize this process. 9 However, these tools are not panaceas. Systematic reviews have shown that available assessment tools do not reliably predict future self-harm or suicide and often perform well on either sensitivity or specificity, but rarely both. 8 This trade-off has significant consequences: low sensitivity leads to missed cases (false negatives), compromising patient safety, while low specificity leads to over-identification (false positives), which can strain already limited mental health resources. 8 Furthermore, research has demonstrated that single-item screens or the use of depression screening as a proxy for suicide risk is insufficient, as these methods can miss a substantial portion of at-risk individuals who may not present with classic depressive symptoms. 13 The Emergence of AI as a Potential Adjunct in Psychiatric Triage The convergence of the adolescent mental health crisis and the limitations of current assessment methods creates a critical need for innovative solutions. Artificial intelligence (AI), and specifically Large Language Models (LLMs), have emerged as powerful technologies capable of processing vast amounts of unstructured text data to identify patterns and support clinical decision-making. 15 The potential applications in mental health are compelling. AI could help address workforce shortages by automating administrative tasks, freeing clinicians to focus on patient care. 18 LLMs, trained on massive text corpora, can analyze clinical notes, patient communications, and social media data to detect signs of distress or risk. 20 Early studies have shown promise in using LLMs to detect suicidal ideation from text, suggesting a potential role in risk prediction and triage. 22 Study Rationale, Objectives, and Hypotheses This technological promise presents a critical paradox of scalability versus safety. While LLMs offer an unprecedented ability to scale mental health screening, this also creates the risk of deploying flawed or biased systems at scale, potentially causing widespread harm. 25 There remains a significant gap in the literature regarding the rigorous, comparative evaluation of modern LLMs for the specific, high-stakes task of adolescent suicide risk triage. Critically, evaluation must extend beyond simple classification accuracy to assess the clinical appropriateness and safety of the actions these models recommend. This study was designed to address this gap. The primary objective was to evaluate and compare the performance of three prominent LLMs (GPT-4o, Claude 3.5 Sonnet, and Llama-3.1-70B) against a clinical gold standard in two domains: (1) classifying the level of suicide risk, and (2) generating clinically appropriate and safe action plans. The evaluation was conducted using a set of purpose-built synthetic adolescent psychiatric vignettes, a methodology that allows for controlled and systematic testing without the ethical and privacy constraints of using real patient data. We formulated three hypotheses: (1) LLM performance in risk classification would vary, with the more advanced proprietary models (GPT-4o, Claude 3.5 Sonnet) outperforming the open-weight model (Llama-3.1-70B); (2) all models would demonstrate moderate-to-high accuracy in classifying cases at the extremes of risk (i.e., no risk versus very high risk) but would struggle with more nuanced, intermediate-risk presentations; and (3) a qualitative analysis of the LLM-generated action plans would reveal significant deficiencies in clinical reasoning and safety planning when compared to the gold standard established by expert clinicians. Methods Study Design This study employed a comparative, in-silico evaluation design. This approach involves testing computational models on standardized, synthetic data to assess performance in a controlled environment. 27 This design was selected to enable the systematic manipulation of clinical variables within the vignettes and to allow for a direct comparison of LLM outputs against a predefined gold standard, thereby avoiding the ethical complexities and regulatory requirements associated with using real patient data. 28 A study flow diagram, modeled after the CONSORT statement, illustrates the process from vignette development to final data analysis ( Figure 1 ). 28 Download figure Open in new tab Figure 1. Study Flow Diagram Flow diagram showing the progression from vignette creation, LLM processing, to data analysis. Synthetic Vignette Development and Validation A set of 40 synthetic clinical vignettes was developed, adhering to established best practices for vignette methodology in health research. 29 Vignettes are a recognized and cost-effective method for assessing clinical decision-making and exploring variations in practice. 30 Each vignette was structured using the “4 P’s” of clinical formulation to ensure clinical depth and realism: P redisposing factors (e.g., family history of suicide, chronic illness), P recipitating factors (e.g., academic pressure, recent interpersonal conflict), P erpetuating factors (e.g., ongoing bullying, substance use, social isolation), and P rotective factors (e.g., strong family support, engagement in extracurricular activities, future-oriented thinking). 32 To ensure a comprehensive test set, variables within the vignettes were systematically controlled and manipulated. 33 Controlled variables included the patient’s age range (13–17 years), the clinical setting (e.g., emergency department, outpatient pediatric clinic, school counselor’s office), and the narrative perspective (e.g., a third-person psychiatric consult request, a first-person account from a concerned parent). Key experimentally manipulated variables included the severity of suicidal ideation (from passive wishes to be dead to active ideation with a specific plan and intent), history of prior suicide attempts, presence and nature of non-suicidal self-injury (NSSI), and the co-occurrence of significant risk factors such as substance use disorders, psychosis, impulsivity, and access to lethal means. 7 The full set of vignettes underwent a content validation process, wherein a panel of two board-certified child and adolescent psychiatrists independently reviewed each case for clinical plausibility, realism, clarity, and internal consistency. 30 Any disagreements were resolved through consensus discussion. The distribution of key clinical characteristics across the final set of 40 vignettes is detailed in Table 1 . View this table: View inline View popup Table 1. Characteristics of the 40 Synthetic Clinical Vignettes Gold Standard Definition For each of the 40 vignettes, the expert clinical panel established a two-part “gold standard.” First, a definitive risk classification was assigned using the hierarchical structure of the Columbia-Suicide Severity Rating Scale (C-SSRS). 11 This scale provides clear, operationalized definitions for different levels of suicidal ideation and behavior, ranging from a passive wish to be dead to an active attempt. Second, the panel formulated a corresponding gold-standard clinical action plan for each vignette. These plans were based on established clinical practice guidelines from the American Psychiatric Association (APA) and the framework of the Suicide Assessment Five-Step Evaluation and Triage (SAFE-T) Pocket Card. 7 The action plans were stratified by risk level and specified necessary interventions, such as ensuring patient safety (e.g., initiating 1:1 observation), determining the level of care (e.g., recommending immediate psychiatric hospitalization), and providing specific treatment recommendations (e.g., conducting means restriction counseling with the family). Large Language Model Selection and Configuration Three state-of-the-art LLMs were selected to represent the current landscape of AI technology: GPT-4o (a leading proprietary model from OpenAI), Claude 3.5 Sonnet (a proprietary model from Anthropic noted for its advanced reasoning capabilities), and Llama-3.1-70B (a powerful open-weight model from Meta). 37 This selection allows for a comparison across different model architectures and training philosophies. A structured, multi-step prompt was engineered to guide the LLMs’ analytical process, moving beyond simple classification to elicit a form of transparent reasoning. 40 This “chain-of-thought” (CoT) prompting technique is designed to break down a complex task into intermediate steps, improving performance and providing insight into the model’s decision-making process. 41 The prompt given to each LLM for every vignette consisted of four parts: Role Assignment “You are an expert child and adolescent psychiatrist specializing in suicide risk assessment.” 42 Context and Task Definition “You will be provided with a clinical vignette describing an adolescent patient. Your task is to conduct a comprehensive suicide risk assessment and formulate a detailed, step-by-step clinical action plan.” 43 Structured Reasoning Steps (CoT) “To complete this task, you must proceed in the following order. First, analyze the vignette and explicitly list all identifiable risk factors and protective factors. Second, based on your analysis, classify the patient’s current suicide risk level using the specific categories of the Columbia-Suicide Severity Rating Scale (C-SSRS). Third, based on your risk classification, provide a clear, actionable, step-by-step clinical action plan.” 40 Output Formatting “Structure your final response with the following distinct headings: ‘Identified Risk and Protective Factors’, ‘C-SSRS Risk Classification’, and ‘Clinical Action Plan’.” Each of the 40 vignettes was submitted as input to each of the three LLMs. To ensure reproducibility while allowing for coherent text generation, a consistent temperature setting of was used for all model queries. 44 All prompts and the corresponding full-text outputs were logged for subsequent analysis. Ethical Considerations This study exclusively utilized synthetically generated data that contained no personal or identifiable information of any real individuals. As such, the research did not involve human participants, and according to institutional and international guidelines, did not require review and approval from an ethics committee or institutional review board. Outcome Measures and Analysis The evaluation of the LLM outputs was conducted through a two-pronged approach. First, a quantitative analysis was performed on the risk classification component. The C-SSRS classification generated by each LLM was compared against the expert-defined gold standard for each vignette. Standard performance metrics for multi-class classification were calculated: overall accuracy, and class-specific precision, recall, and F1-score. 39 A confusion matrix was also generated for each model to provide a visual representation of misclassification patterns. Second, a qualitative analysis was conducted on the LLM-generated clinical action plans. This analysis is critical, as the appropriateness of the recommended actions is arguably more clinically significant than the classification label itself. A rigorous thematic analysis was performed on the 120 generated action plans (40 vignettes x 3 LLMs). 46 Two independent researchers (a clinical psychologist and a psychiatric resident), blind to the LLM that generated each plan, coded the content using a hybrid deductive-inductive approach. The deductive coding framework was based on established safety guidelines 7 and evaluated the plans across three core domains: (1) Safety Interventions (e.g., presence of recommendations for immediate safety measures like removing access to means or ensuring constant observation); (2) Assessment Comprehensiveness (e.g., inclusion of steps like involving family/caregivers, conducting a full psychiatric evaluation); and (3) Disposition Appropriateness (e.g., proportionality of the recommended level of care to the identified risk). An inductive approach was used concurrently to identify emergent themes not captured by the deductive framework, such as the generation of inappropriate therapeutic language or model “hallucinations”. 37 Inter-rater reliability (Cohen’s Kappa) was calculated for the deductive codes, with any discrepancies resolved through discussion with a third senior researcher (Author B). Results Study Flow The study process is summarized in Figure 1 . A total of 40 unique synthetic vignettes were created and validated. Each vignette was processed by the three selected LLMs, resulting in 120 unique LLM outputs. All 120 outputs were included in the final quantitative and qualitative analyses. Quantitative Performance in Risk Classification The overall performance of the three LLMs in classifying suicide risk according to the C-SSRS framework varied considerably. GPT-4o demonstrated the highest performance across all metrics, followed by Claude 3.5 Sonnet, with Llama-3.1-70B showing the lowest performance. Detailed metrics are presented in Table 2 . View this table: View inline View popup Download powerpoint Table 2. Comparative Performance of LLMs in C-SSRS Risk Classification While overall accuracy provides a general benchmark, the F1-scores for specific risk categories revealed more nuanced challenges. All models performed well in distinguishing between no/low risk (C-SSRS levels 1-2) and high risk (C-SSRS level 5 and recent attempt). However, performance dropped significantly for intermediate categories. For instance, GPT-4o’s F1-score for correctly identifying ‘Active Ideation with Some Intent to Act, without Specific Plan’ (C-SSRS level 4) was only 0.65. This indicates a systematic difficulty in distinguishing this critical level of risk, where intent is present but a plan is not yet fully formed, from adjacent categories. Analysis of confusion matrices (see Supplementary Material) confirmed that the most common errors involved misclassifying level 4 cases as either level 3 (method without intent) or level 5 (plan with intent). Qualitative Assessment of Clinical Action Plans The thematic analysis of the 120 generated action plans revealed several consistent patterns across all three models. While the models demonstrated a basic capacity to recommend standard safety measures, they exhibited critical deficiencies in nuanced clinical reasoning and comprehensive safety planning. The key themes are summarized in Table 3 and elaborated below. View this table: View inline View popup Download powerpoint Table 3. Qualitative Thematic Analysis of LLM-Generated Clinical Action Plans Theme 1: Adherence to Basic Safety Protocols In cases with clear, high-risk features (e.g., a specific suicide plan with intent, a recent attempt), all three LLMs reliably generated appropriate high-level recommendations. These included advising immediate transport to an emergency department, recommending psychiatric hospitalization, and stating the need for a full psychiatric evaluation. This suggests the models have been trained on data that effectively links explicit high-risk keywords to standard emergency protocols. Theme 2: Omission of Critical Inquiries and Actions This was the most concerning and pervasive failure. Even when correctly identifying high risk and recommending hospitalization, the models’ action plans frequently lacked crucial, specific steps that are standard clinical practice. The most significant omission was the failure to explicitly prompt the user to inquire about the patient’s access to the identified suicide method. For example, a model might note a plan involving overdose but not include the action: “Question the patient and parents about what medications are in the home and advise on securing or removing them.” This represents a critical failure in means restriction counseling, a cornerstone of suicide prevention. Theme 3: Generation of Inappropriate Reassurance or Therapeutic Language The models often blurred the line between assessment/triage and therapy. Outputs frequently included statements of empathy and reassurance (e.g., “It’s brave of you to share this,” “Things can get better”) that, while well-intentioned, are inappropriate and potentially counterproductive in an initial safety assessment where the focus must be on immediate risk and disposition.42 This suggests a model bias toward a generic “supportive chatbot” persona, which conflicts with the required objectivity of a clinical triage tool. Theme 4: Inconsistent Identification and Use of Protective Factors While the prompt explicitly asked for the identification of protective factors, the models’ ability to do so was inconsistent. More importantly, even when protective factors were correctly identified, they were rarely integrated into the proposed action plan. For instance, a vignette might describe a teen who is alienated from their parents but has a strong bond with an aunt. A robust clinical plan would leverage this by suggesting the aunt be involved in the safety planning process. The LLMs consistently failed to make this connection, treating the identification of protective factors as a separate task rather than an integral part of intervention planning. Theme 5: Reasoning-Action Mismatch The analysis of the CoT outputs revealed a troubling disconnect between the models’ “reasoning” and their “planning.” In several instances, a model would accurately list numerous severe risk factors (e.g., prior attempt, substance use, access to means), provide a correct high-risk C-SSRS classification, and then inexplicably generate a low-intensity action plan (e.g., “recommend weekly therapy”). This suggests a failure to properly weight the identified factors and translate the assessed risk level into a proportionate clinical action, pointing to a fundamental weakness in their simulated clinical judgment. Discussion Summary of Principal Findings This study provides a multi-faceted evaluation of the capabilities of three leading LLMs in the critical psychiatric task of adolescent suicide risk triage. The findings present a mixed but cautionary picture. Quantitatively, the models demonstrated a rudimentary ability to classify risk, with performance correlating with model sophistication. However, even the best-performing model, GPT-4o, struggled to accurately classify nuanced, intermediate-risk cases, which often represent a crucial window for clinical intervention. More critically, the qualitative analysis of the recommended action plans revealed profound and consistent deficiencies. Across all models, there was a pattern of omitting essential safety actions, particularly around means restriction, and a failure to integrate protective factors into planning. The generation of clinically inappropriate therapeutic language and instances of a stark mismatch between risk assessment and action planning further underscore the superficial nature of their current “clinical reasoning” capabilities. Interpretation and Comparison with Existing Literature Our quantitative results are broadly consistent with a growing body of literature demonstrating that LLMs can perform clinical classification tasks with a degree of accuracy, sometimes approaching that of human non-experts. 22 However, this study’s primary contribution lies in its deep qualitative analysis of the actionable outputs , a less-explored but arguably more important domain for patient safety. This is where the models’ limitations become most apparent. The failure to move from pattern recognition (classifying risk) to sound clinical judgment (creating a safe, comprehensive plan) highlights a critical gap. A crucial interpretation of these findings relates to the use of synthetic data. While this methodology enables controlled evaluation, it also represents an idealized, best-case scenario. 27 The vignettes used, though designed for realism, are inherently clean, well-structured narratives containing all relevant information. Real-world clinical presentations are invariably more complex, fragmented, ambiguous, and laden with confounding factors. 31 The fact that the LLMs exhibited significant failure modes even on this simplified data strongly suggests that their performance in a real clinical environment would be substantially worse. The observed omissions and reasoning errors are likely to be amplified when faced with the “messy” data of actual clinical practice. Clinical and Practical Implications The results of this study directly inform the ongoing debate about the role of AI in mental healthcare. The central tension between the promise of scalability and the imperative of safety is starkly illustrated. The potential to deploy an LLM-based tool to screen millions of individuals is alluring, especially given provider shortages. 19 However, our findings suggest that deploying the current generation of LLMs as autonomous triage agents would be unsafe and irresponsible. The risk of scaled harm—for instance, systematically failing to advise on means restriction for thousands of at-risk adolescents—is unacceptably high. Instead, a more viable path forward may lie in conceptualizing these tools not as autonomous decision-makers, but as “clinical co-pilots” or decision support systems designed to augment, not replace, human clinicians. 49 For example, in a primary care or school setting where a non-specialist is conducting an initial screening, an AI tool could be used to structure the assessment. Rather than providing a definitive risk level or plan, the tool could prompt the human user with a checklist of critical questions derived from the patient’s narrative (e.g., “The patient mentioned feeling hopeless. Have you asked about suicidal thoughts? The patient mentioned a plan involving pills. Have you asked about access to medications in the home?”). This human-in-the-loop model leverages the AI’s data processing strengths while keeping clinical judgment and responsibility firmly in the hands of a trained professional, mitigating risks associated with automation bias and alert fatigue. 49 Ethical Considerations and Algorithmic Bias The potential deployment of such technology raises profound ethical questions. The issue of accountability is paramount. If an AI-assisted triage decision contributes to an adverse outcome, where does liability lie? With the clinician who used the tool, the healthcare system that deployed it, or the developer who created the model? Current legal and professional frameworks are ill-equipped to address this new technological reality. 52 Furthermore, the risk of algorithmic bias is a major concern. While this study used synthetic data that did not include demographic characteristics, it is well-documented that AI models trained on real-world data can absorb and amplify existing societal biases. 25 Studies have already shown racial bias in AI-generated treatment recommendations for psychiatric disorders. 26 Deploying a suicide risk model trained on biased data could lead to the systematic under-or over-estimation of risk for certain populations, exacerbating existing health disparities. Finally, the principle of transparency is essential for building trust and ensuring safety. The “black box” nature of many LLMs is a significant barrier to clinical adoption. Our use of chain-of-thought prompting was a deliberate attempt to foster interpretability, but true transparency in how these complex models arrive at their conclusions remains an unsolved technical and ethical challenge. 55 Limitations of the Study This study has several important limitations. The foremost is its reliance on synthetic data, which, as discussed, limits the generalizability of the findings to complex, real-world clinical encounters. Second, our evaluation was based on a single, albeit carefully engineered, prompt structure. The performance of LLMs is known to be highly sensitive to prompt design, and different prompting strategies might yield different results. Third, the “gold standard” for risk and action plans was established by a small panel of two experts; while both are highly experienced, clinical judgment can vary among professionals. Lastly, the field of AI is evolving at an extraordinary pace. The specific model versions tested here (GPT-4o, Claude 3.5 Sonnet, Llama-3.1-70B) will inevitably be superseded. However, the evaluation framework and the types of errors identified are likely to remain relevant for assessing future generations of LLMs. Future Directions This research highlights several critical avenues for future work. There is an urgent need to validate these findings using large-scale, de-identified real-world data, such as electronic health records or crisis text line transcripts, under strict ethical and privacy-preserving protocols. Such studies would provide a more realistic assessment of LLM performance. Concurrently, the field must move beyond simple accuracy metrics to develop and validate more sophisticated evaluation frameworks that measure clinical utility, safety, and equity. 56 Finally, research in human-computer interaction is needed to explore how AI-generated insights can be presented to clinicians in a way that genuinely supports decision-making without inducing automation bias or overwhelming users with low-value alerts. Conclusion This comparative evaluation of leading LLMs for adolescent suicide risk triage offers a sobering yet crucial perspective on the current state of AI in psychiatry. While these models demonstrate a nascent potential for processing clinical narratives and recognizing basic risk patterns, they concurrently exhibit fundamental flaws in clinical reasoning, safety planning, and the generation of appropriate, actionable guidance. They lack the nuanced judgment, contextual awareness, and ethical grounding required for a high-stakes, life-or-death clinical task. The findings strongly suggest that current-generation LLMs are not ready for autonomous clinical deployment in this domain. The path to responsibly integrating AI into suicide prevention must be guided by an unwavering commitment to patient safety, demanding rigorous, clinically-grounded validation, the development of robust ethical and regulatory guardrails, and the preservation of human expertise as the ultimate arbiter of patient care. Data Availability All data produced in the present study are available upon reasonable request to the authors Declarations Ethics approval and consent to participate Not applicable. This study was conducted using exclusively synthetic, non-human data and did not involve human participants. Consent for publication Not applicable. Availability of data and materials The set of 40 synthetic clinical vignettes generated for this study and the analysis code are available from the corresponding author upon reasonable request, in accordance with the journal’s data sharing policies. Competing interests The authors declare that they have no competing interests. Funding This research received no specific grant from any funding agency in the public, commercial, or not-for-profit sectors. Authors’ contributions Author M.M. conceived the study, designed the methodology, conducted the LLM queries and quantitative analysis, and drafted the manuscript. Author M.M. developed the synthetic vignettes, established the clinical gold standard, participated in the qualitative analysis, and critically revised the manuscript for clinical content. Author M.M. supervised the project, co-developed the study design and methodology, resolved discrepancies in the qualitative analysis, and critically revised the manuscript for intellectual content. Author has read and approved the final manuscript. Acknowledgements None. References 1. ↵ World Health Organization . Suicide . Published September 12, 2024 . Accessed August 5, 2025 . https://www.who.int/news-room/fact-sheets/detail/suicide 2. ↵ American Foundation for Suicide Prevention . Suicide statistics . Published 2025. Accessed August 5, 2025 . https://afsp.org/suicide-statistics/ 3. Centers for Disease Control and Prevention, National Center for Health Statistics . National Vital Statistics System – Mortality data (2023) via CDC WONDER . Accessed August 5 , 2025 . 4. ↵ Centers for Disease Control and Prevention . Data on Suicidal Thoughts and Behavior Among US Youth . Updated July 2025. Accessed August 5, 2025 . https://www.cdc.gov/mental-health/about-data/suicidal-thoughts-and-behavior.html 5. ↵ Nath R , Matthews DD , DeChants JP , et al. 2024 U.S. National Survey on the Mental Health of LGBTQ+ Young People . The Trevor Project ; 2024 . 6. ↵ The Jed Foundation . Mental Health and Suicide Statistics . Updated May 2025. Accessed August 5, 2025 . https://jedfoundation.org/mental-health-and-suicide-statistics/ 7. ↵ Tofte J , Bahar M , Bipeta R , et al. Suicide Risk Assessment . Focus (Am Psychiatr Publ) . 2020 ; 18 ( 2 ): 126 – 134 . OpenUrl 8. ↵ Large R , Ryan C. How accurate are suicide risk prediction models? Asking the right questions for clinical practice . BMJ Ment Health . 2021 ; 24 ( 1 ): e1 – e4 . OpenUrl Abstract / FREE Full Text 9. ↵ Horowitz LM , Bridge JA , Pao M , et al. Screening youth for suicide risk in the emergency department: The Ask Suicide-Screening Questions (ASQ) . Arch Pediatr Adolesc Med . 2012 ; 166 ( 12 ): 1170 – 1176 . OpenUrl CrossRef PubMed 10. Posner K , Brown GK , Stanley B , et al. The Columbia-Suicide Severity Rating Scale: initial validity and internal consistency findings from a prospective multisite study . Am J Psychiatry . 2011 ; 168 ( 12 ): 1266 – 1277 . OpenUrl CrossRef PubMed Web of Science 11. ↵ Columbia University Department of Psychiatry . The Columbia Protocol (C-SSRS) . Accessed August 5, 2025 . https://cssrs.columbia.edu/ 12. Quinlivan L , Cooper J , Davies L , et al. Which are the most useful scales for predicting repeat self-harm? A systematic review of 27 scales . J Affect Disord . 2017 ; 211 : 128 – 136 . OpenUrl 13. ↵ Martin A , Ludi E , King CA , et al. Is Depression Screening a Sufficient Proxy for Suicide Risk Screening in Medically Ill Hospitalized Patients? J Gen Intern Med . 2021 ; 36 ( 8 ): 2333 – 2339 . OpenUrl 14. Hom MA , Stanley IH , Joiner TE Jr . . Evaluating the association between lifetime suicide attempts and current suicidal ideation: The moderating role of hopelessness and thwarted belongingness . J Affect Disord . 2015 ; 186 : 136 – 143 . OpenUrl 15. ↵ Abd-Alrazaq A , Alajlani M , Alhuwail D , et al. Artificial intelligence in mental health care: a systematic review of diagnosis, monitoring, and intervention applications . Psychol Med . 2022 ; 52 ( 16 ): 3211 – 3222 . OpenUrl 16. Blease C , Kharko A , Bernstein M. Artificial Intelligence and the Future of Psychiatry: Insights from a Global Survey of Psychiatrists . J Med Internet Res . 2021 ; 23 ( 1 ): e23432 . OpenUrl 17. Graham S , Depp C , Lee EE , et al. Artificial Intelligence for Mental Health and Mental Illnesses: an Overview . Curr Psychiatry Rep . 2019 ; 21 ( 11 ): 116 . OpenUrl CrossRef PubMed 18. ↵ Torous J , Jänicke M , Stern AD . A new era for digital mental health: the arrival of large language models . Lancet Digit Health . 2023 ; 5 ( 6 ): e332 – e333 . OpenUrl 19. ↵ Lee EE , Torous J , De Choudhury M , et al. The Future of AI in Psychiatry: A Report from the APA/APPI Work Group on AI . Am J Psychiatry . 2025 ; 182 ( 1 ): 12 – 24 . OpenUrl 20. ↵ Coppersmith G , Leary R , Crutchley P , et al. Natural language processing for the assessment of suicide risk . Behav Res Methods . 2018 ; 50 ( 5 ): 1965 – 1981 . OpenUrl 21. Zirikli A , Fodeh S , Al-Garadi MA , et al. A systematic review of the applications of artificial intelligence in suicide and self-harm research . J Affect Disord . 2023 ; 323 : 718 – 731 . OpenUrl 22. ↵ Levkovich I. Suicide Risk Assessments Through the Eyes of ChatGPT-3.5 Versus ChatGPT-4: Vignette Study . JMIR Ment Health . 2023 ; 10 : e50729 . OpenUrl 23. Aladağ AE , Muderrisoglu S , Akbas NB , et al. Detecting Suicidal Ideation on Forums: Proof-of-Concept Study . J Med Internet Res . 2018 ; 20 ( 6 ): e215 . OpenUrl PubMed 24. Lee C , Raghuram S , Raghu V , et al. Crisis prediction among tele-mental health patients: A large language model and expert clinician comparison . JMIR Ment Health . 2024 ; 11 : e55766 . OpenUrl 25. ↵ Reddy S , Allan S , Coghlan S , et al. A governance model for the application of AI in health care . J Am Med Inform Assoc . 2020 ; 27 ( 3 ): 491 – 497 . OpenUrl CrossRef PubMed 26. ↵ Obermeyer Z , Powers B , Vogeli C , et al. Dissecting racial bias in an algorithm used to manage the health of populations . Science . 2019 ; 366 ( 6464 ): 447 – 453 . OpenUrl Abstract / FREE Full Text 27. ↵ Viceconti M , Henney A , Morley-Fletcher E. In silico trials: verification, validation and uncertainty quantification of predictive models used in the regulatory submission of biomedical products . Interface Focus . 2021 ; 11 ( 2 ): 20200089 . OpenUrl 28. ↵ Chen RJ , Lu MY , Chen TY , et al. Synthetic data in machine learning for medicine and healthcare . Nat Biomed Eng . 2021 ; 5 ( 6 ): 493 – 497 . OpenUrl PubMed 29. ↵ Peabody JW , Luck J , Glassman P , et al. Comparison of vignettes, standardized patients, and chart abstraction: a prospective validation study of 3 methods for measuring quality . JAMA . 2000 ; 283 ( 13 ): 1715 – 1722 . OpenUrl CrossRef PubMed Web of Science 30. ↵ Evans SC , Roberts MC , Keeley JW , et al. Vignette-based methodologies for studying clinicians’ decision-making: validity, utility, and application in ICD-11 field studies . Int J Clin Health Psychol . 2015 ; 15 ( 2 ): 160 – 170 . OpenUrl CrossRef PubMed 31. ↵ Bachmann LM , Mühleisen A , Bock A , et al. The clinical vignette in internal medicine: a narrative review and practical guide . Swiss Med Wkly . 2008 ; 138 ( 23-24 ): 336 – 340 . OpenUrl 32. ↵ Johnstone K , Whitton A. Clinical formulation: where it came from, what it is, and why it matters . BJPsych Advances . 2021 ; 27 ( 2 ): 82 – 91 . OpenUrl 33. ↵ Atzmüller C , Steiner PM . Experimental vignette studies in survey research . Methodology . 2010 ; 6 ( 3 ): 128 – 138 . OpenUrl 34. U.S. Department of Veterans Affairs; U.S. Department of Defense . VA/DoD Clinical Practice Guideline for the Assessment and Management of Patients at Risk for Suicide . 2019 . 35. Posner K , Brown GK , Stanley B , et al. The Columbia-Suicide Severity Rating Scale: initial validity and internal consistency findings from a prospective multisite study . Am J Psychiatry . 2011 ; 168 ( 12 ): 1266 – 1277 . OpenUrl CrossRef PubMed Web of Science 36. Substance Abuse and Mental Health Services Administration . Suicide Assessment Five-Step Evaluation and Triage (SAFE-T) Pocket Card for Clinicians. HHS Publication No. (SMA) 09-4432 . Rockville, MD : Center for Mental Health Services, SAMHSA ; 2009 . 37. ↵ OpenAI . GPT-4o Technical Report . 2024 . Accessed August 5, 2025 . https://openai.com/index/hello-gpt-4o/ 38. Anthropic . Introducing the next generation of Claude . 2024 . Accessed August 5, 2025 . https://www.anthropic.com/news/claude-3-5-sonnet 39. ↵ Meta AI . Introducing Meta Llama 3.1 . 2024 . Accessed August 5, 2025 . https://ai.meta.com/blog/meta-llama-3-1/ 40. ↵ Wei J , Wang X , Schuurmans D , et al. Chain-of-thought prompting elicits reasoning in large language models . Adv Neural Inf Process Syst . 2022 ; 35 : 24824 – 24837 . OpenUrl 41. ↵ Kojima T , Gu SS , Reid M , et al. Large language models are zero-shot reasoners . Adv Neural Inf Process Syst . 2022 ; 35 : 22199 – 22213 . OpenUrl 42. ↵ White J , Fu Q , Hays S , et al. A prompt pattern catalog to enhance prompt engineering with ChatGPT . arXiv preprint arxiv: 2302.11382 . 2023 . 43. ↵ Brown TB , Mann B , Ryder N , et al. Language models are few-shot learners . Adv Neural Inf Process Syst . 2020 ; 33 : 1877 – 1901 . OpenUrl 44. ↵ Holtzman A , Buys J , Du L , et al. The curious case of neural text degeneration . In: Proceedings of the International Conference on Learning Representations . 2020 . 45. Fawcett T. An introduction to ROC analysis . Pattern Recognit Lett . 2006 ; 27 ( 8 ): 861 – 874 . OpenUrl CrossRef Web of Science 46. ↵ Braun V , Clarke V. Using thematic analysis in psychology . Qual Res Psychol . 2006 ; 3 ( 2 ): 77 – 101 . OpenUrl CrossRef 47. Kiger ME , Varpio L. Thematic analysis of qualitative data: AMEE Guide No. 131 . Med Teach . 2020 ; 42 ( 8 ): 846 – 854 . OpenUrl CrossRef PubMed 48. Fiske A , Henningsen N , Buyx A. Your robot therapist will see you now: ethical implications of embodied artificial intelligence in psychiatry, psychology, and psychotherapy . J Med Internet Res . 2019 ; 21 ( 5 ): e13216 . OpenUrl CrossRef PubMed 49. ↵ Topol EJ . High-performance medicine: the convergence of human and artificial intelligence . Nat Med . 2019 ; 25 ( 1 ): 44 – 56 . OpenUrl CrossRef PubMed 50. Upadhyay R. AI as a Copilot in Healthcare: Enhancing, Not Replacing, Clinical Decision-Making . J Comput Sci Technol Stud . 2025 ; 7 ( 8 ): 08 – 14 . OpenUrl 51. Goddard K , Roudsari A , Wyatt JC . Automation bias: a systematic review of frequency, effect mediators, and mitigators . J Am Med Inform Assoc . 2012 ; 19 ( 1 ): 121 – 127 . OpenUrl CrossRef PubMed 52. ↵ Price WN II , Cohen IG . Privacy in the age of medical big data . Nat Med . 2019 ; 25 ( 1 ): 37 – 43 . OpenUrl CrossRef PubMed 53. Matheny ME , Thadaney Israni S , Ahmed M , et al. Artificial intelligence in health care: the hope, the hype, the promise, the peril . NAM Special Publication . 2019 . 54. Wiens J , Saria S , Sendak M , et al. Do no harm: a roadmap for responsible machine learning for health care . Nat Med . 2019 ; 25 ( 9 ): 1337 – 1340 . OpenUrl CrossRef PubMed 55. ↵ Vayena E , Blasimme A , Cohen IG . Machine learning in medicine: addressing ethical challenges . PLoS Med . 2018 ; 15 ( 11 ): e1002689 . OpenUrl CrossRef PubMed 56. ↵ Liu Y , Chen PC , Krause J , et al. How to read articles that use machine learning: users’ guides to the medical literature . JAMA . 2019 ; 322 ( 18 ): 1806 – 1816 . OpenUrl CrossRef PubMed 57. Collins GS , Reitsma JB , Altman DG , et al. Transparent reporting of a multivariable prediction model for individual prognosis or diagnosis (TRIPOD): the TRIPOD statement . Ann Intern Med . 2015 ; 162 ( 1 ): 55 – 63 . OpenUrl CrossRef PubMed 58. Vollmer S , Mateen BA , Bohner G , et al. Machine learning and artificial intelligence research for patient benefit: 20 critical questions on transparency, replicability, ethics, and effectiveness . BMJ . 2020 ; 368 : l6927 . OpenUrl FREE Full Text View the discussion thread. Back to top Previous Next Posted August 07, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following AI-Powered Triage of Suicidal Ideation in Adolescents: A Comparative Evaluation of Large Language Models Using Synthetic Clinical Vignettes Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share AI-Powered Triage of Suicidal Ideation in Adolescents: A Comparative Evaluation of Large Language Models Using Synthetic Clinical Vignettes Masab Mansoor medRxiv 2025.08.05.25333046; doi: https://doi.org/10.1101/2025.08.05.25333046 Share This Article: Copy Citation Tools AI-Powered Triage of Suicidal Ideation in Adolescents: A Comparative Evaluation of Large Language Models Using Synthetic Clinical Vignettes Masab Mansoor medRxiv 2025.08.05.25333046; doi: https://doi.org/10.1101/2025.08.05.25333046 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Psychiatry and Clinical Psychology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (541) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a008df599db5e2c5',t:'MTc3OTU4OTc2NA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.