Full text
47,954 characters
· extracted from
preprint-html
· click to expand
Large Language Models in Stroke Management: A Review of the Literature | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Large Language Models in Stroke Management: A Review of the Literature View ORCID Profile Shelly Soffer , View ORCID Profile Aya Mudrik , View ORCID Profile Orly Efros , View ORCID Profile Mahmud Omar , View ORCID Profile Girish N Nadkarni , View ORCID Profile Eyal Klang doi: https://doi.org/10.1101/2025.06.28.25330477 Shelly Soffer 1 Institute of Hematology, Davidoff Cancer Center, Rabin Medical Center , Petah-Tikva, Israel 2 Gray Faculty of Medical and Health Science, Tel Aviv University , Tel Aviv, Israel Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Shelly Soffer For correspondence: soffer.shelly{at}gmail.com Aya Mudrik 3 Ben-Gurion University of the Negev , Be’er Sheva, Israel Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Aya Mudrik Orly Efros 2 Gray Faculty of Medical and Health Science, Tel Aviv University , Tel Aviv, Israel 4 National Hemophilia Center and Institute of Thrombosis & Hemostasis, Chaim Sheba Medical Center , Tel Hashomer, Israel Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Orly Efros Mahmud Omar 5 The Windreich Department of Artificial Intelligence and Human Health, Mount Sinai Medical Center , NY, USA 6 The Division of Data-Driven and Digital Medicine (D3M), Icahn School of Medicine at Mount Sinai , New York, NY, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Mahmud Omar Girish N Nadkarni 5 The Windreich Department of Artificial Intelligence and Human Health, Mount Sinai Medical Center , NY, USA 6 The Division of Data-Driven and Digital Medicine (D3M), Icahn School of Medicine at Mount Sinai , New York, NY, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Girish N Nadkarni Eyal Klang 5 The Windreich Department of Artificial Intelligence and Human Health, Mount Sinai Medical Center , NY, USA 6 The Division of Data-Driven and Digital Medicine (D3M), Icahn School of Medicine at Mount Sinai , New York, NY, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Eyal Klang Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Stroke care generates vast free-text records that slow chart review and hamper data reuse. Large language models (LLMs) have been trialed as a remedy in tasks ranging from imaging interpretation to outcome prediction. To assess current applications of LLMs in stroke management, we conducted a narrative review by searching PubMed and Google Scholar databases on January 30, 2025, using stroke- and LLM-related terms. This review included fifteen studies demonstrating that LLMs can: (i) extract key variables from thrombectomy reports with up to 94% accuracy, (ii) localize stroke lesions from case-report text with F1 scores of 0.74–0.85, and (iii) forecast functional outcome more accurately than legacy bedside scores in small pilot cohorts. These results, however, rest on narrow, retrospective datasets-often from single centers or publicly available case reports that the models may have encountered during pre-training. Most evaluations use proprietary systems, limiting reproducibility and obscuring prompt design. None stratify performance by sex, language, or socioeconomic status, and few disclose safeguards against hallucination or data leakage. We conclude that LLMs are credible research tools for text mining and hypothesis generation in stroke, but evidence for clinical deployment remains preliminary. Rigorous, multisite validation, open benchmarks, bias audits, and human-in-the-loop workflows are prerequisites before LLMs can reliably support time-critical decisions such as thrombolysis or thrombectomy triage. INTRODUCTION Modern stroke care spans from symptom recognition and imaging to endovascular procedures and rehabilitation. 1 This range produces large amounts of clinical text: notes, radiology reports, procedure logs, and discharge summaries. 2 Large language models (LLMs) can integrate these data at scale. They may reveal clinical associations that remain hidden when documentation is scattered. 3 LLMs may also automate time-consuming charting, giving neurologists more time for direct patient care. 4 Properly trained models may detect high-risk stroke presentations, identify subtle imaging findings in free-text reports, or adapt patient education to diverse linguistic and cultural contexts. 5 – 8 Despite these possibilities, questions remain about how LLMs will perform in varied healthcare settings, how they will handle missing or irregular data, and how clinicians can guard against errors or model “hallucinations”. 9 These concerns are critical in stroke, where decisions often hinge on details that emerge under severe time constraints. 10 In this review we outline current LLM applications in stroke management, as presented in Figure 1 . We begin by clarifying basic LLM terminology, then explore their use across stroke management. We conclude with an assessment of limitations, practical considerations, and technical underpinnings. Download figure Open in new tab Figure 1. Applications of Large Language Models (LLMs) in Stroke Management. AI GLOSSARY OF TERMS To establish a shared foundation for the sections that follow, we provide a concise glossary clarifying key concept such as artificial intelligence (AI), natural language processing (NLP), and large language models (LLMs). Figure 2 presents a hierarchy diagram of AI terms. Download figure Open in new tab Figure 2. A hierarchy diagram of AI terms. Artificial Intelligence (AI) AI encompasses computational methods that perform tasks traditionally requiring human judgment. 9 , 11 In stroke care, AI tools can analyze imaging findings, vital signs, and clinical text. 12 , 13 This accelerates detection of critical events and may improve patient triage. AI requires careful validation to ensure reliable outputs. 14 Machine Learning (ML) and Neural Networks Machine learning is a branch of AI that uses algorithms to discover patterns in data and refine them with training. 15 16 In stroke contexts, ML models can detect subtle indications of early infarction or predict patient outcomes based on risk factors. 12 , 17 Neural networks are a common ML approach, inspired by the organization of neurons in the human brain. They process data through multiple layers of connected nodes. Each “artificial neuron” is similar to one logistic regression unit, weighting inputs from the previous layer, and outputs the result to the next layer. 15 , 18 , 19 In stroke research, such networks can classify CT or MRI findings, and pinpoint factors linked to large-vessel occlusion or hemorrhage. 20 Natural Language Processing (NLP) NLP enables AI systems to interpret and generate human language. It transforms clinical notes— like admission notes or discharge summaries—into analyzable information. 21 , 22 In stroke care, NLP can flag references to symptoms such as “facial droop” or “speech difficulty”. This approach allows rapid, automated extraction of relevant patient details. Clinicians receive concise summaries of key findings, reducing manual chart review. 23 Transformer Architecture and Attention Transformers are an advanced type of neural network that process sequences, such as free-text notes, by assigning weights to different words. This “attention” mechanism highlights important words while preserving context. 24 For example, a transformer may focus on terms like “hemiparesis” or “neglect” in a clinical note. This architecture manages large datasets efficiently. It can prioritize urgent cases by identifying high-risk descriptors in patient histories. In stroke units, where rapid decision-making matters, a transformer’s ability to distill key textual details can improve workflow. Large Language Models (LLMs) LLMs are very large transformer-based models trained on extensive text corpora. They generate context-sensitive responses by predicting the next token in a sequence. In stroke care, an LLM might synthesize patient symptoms and risk factors to suggest possible diagnoses or management steps. Examples of LLMs include closed commercial chatbot models such as openAI’s GPT-4o, 25 GPT-o1, 26 and Anthropic’s Claude-3.5, 27 and open access models such as Meta’s Llama-3.3, 28 and DeepSeek R1. 29 Fine-Tuning Fine-tuning adapts a pre-trained model to a targeted task using additional examples from the desired domain. 30 In stroke research, a general LLM can be fine-tuned with registries of acute ischemic or hemorrhagic stroke cases. This process reduces errors and improves consistency. 31 Prompt Engineering Prompt engineering involves designing specific queries or instructions for an LLM. 32 Precise prompts yield more focused, accurate responses. 33 This technique ensures that AI outputs remain clinically relevant. In practice, a well-constructed prompt can prevent the model from generating off-topic content, saving time in emergency or inpatient settings where clarity is key. 34 , 35 Figure 3 demonstrates a prompt engineering workflow for stroke diagnosis using LLMs. Download figure Open in new tab Figure 3. A prompt engineering workflow for stroke diagnosis using LLMs. Hallucinations An LLM “hallucination” occurs when the model confidently produces content that is factually incorrect or unfounded. 36 Several factors can cause hallucinations, including incomplete training data or ambiguous prompts. Continuous monitoring and human oversight are important. Regular testing of the model’s outputs using real-world scenarios helps maintain reliability. 37 METHODS This review followed general principles for narrative reviews as outlined by Green et al. 38 and Ferrari 39 . We conducted a narrative review to explore the applications of LLMs in the context of stroke. Our literature search was performed on January 30, 2025 using key terms related to both stroke and LLMs across PubMed and Google Scholar. The full list of search terms is provided in the Supplement Materials . In addition, we screened the reference lists of all included articles to identify additional studies. We included studies that addressed the use of LLMs in stroke care, diagnosis, intervention, or research settings, and excluded those that did not substantively engage with both domains. Article selection was based on title and abstract screening, followed by full-text review when appropriate. From the included studies, we extracted information on study design, population characteristics, and reported use cases or outcomes. We organized the findings thematically, emphasizing clinical applications, model functions, and methodological approaches and limitations. Consistent with narrative review methodology, we did not perform formal statistical analyses or risk-of-bias assessments. Instead, we aimed to summarize current developments, map emerging trends, and suggest avenues for future research and clinical practice. CLINICAL APPLICATIONS We included 15 studies in this narrative review. Their characteristics, key findings, and metrics are summarized in Tables 1 – 5 . Limitations and future directions are detailed in Supplementary Tables 1–5 . View this table: View inline View popup Download powerpoint Table 1: Examples of LLM applications in stroke detection and imaging Stroke Detection and Imaging Interpretation Recent investigations highlight how LLMs can assist in stroke detection and imaging interpretation, with direct implications for timely care. For example, Lee et al. evaluated openAI’s GPT-4 40 for localizing acute stroke lesions using raw text from 46 published case reports (BMC Neurology case reports). 41 They asked GPT-4 to determine whether each case involved a single vs. multiple lesion, which brain region (cerebrum, cerebellum, brainstem, or spinal cord), and which side (left, right, or both). Each case was tested three times. Then, GPT-4’s answers were compared with the actual imaging results. Across these 138 trials, GPT-4 achieved F1-scores of about 0.74–0.85 for region/side detection, performing well for cerebral and spinal lesions but struggling more with cerebellar cases. Some errors stemmed from incomplete or ambiguous case descriptions (“extrinsic”), while others reflected GPT-4’s own logical slips (“intrinsic”). One notable limitation is that the cases were published, potentially giving GPT-4 prior exposure to them during its training. Also, case reports often omit routine details and may feature rare presentations, so results may not generalize to everyday clinical notes. LLMs have also demonstrated promise in assisting emergency medical services (EMS) by enhancing prehospital screening tools. The recent retrospective study of 400 emergency department patients by Wang et al. evaluated GPT-4 and GPT-3.5 for the detection of acute ischemic stroke (AIS) and large vessel occlusion (LVO). 42 GPT-4 achieved a higher area under the receiver operating characteristic curve (AUC) for AIS (0.75 vs 0.59) and LVO detection (0.71 vs 0.60), alongside superior sensitivity and specificity for both conditions. Moreover, GPT-4 demonstrated stronger factual correctness (Likert score of 4.24 vs 3.62) and a lower rate of errors (6.8% vs 24.8%) compared to GPT-3.5. Koyun and Taskent retrospectively examined 110 Diffusion-Weighted (DW) MRI cases—55 with AIS and 55 healthy controls—to compare the diagnostic performance of two AI models, openAI’s GPT-4o 25 and Anthropic’s Claude 3.5 Sonnet 27 , in detecting AIS. 7 While GPT-4o achieved an accuracy 51.8%, Claude 3.5 Sonnet showed a substantially higher accuracy 84.5%. In hemisphere-level localization, Claude 3.5 Sonnet was correct in 67.3% of AIS cases, whereas GPT-4o succeeded in 32.7%. For specific region/lobe localization, Claude 3.5 Sonnet’s accuracy reached 30.9%, while GPT-4o’s was 7.3%. A second evaluation two weeks later showed moderate intra-model agreement for both, but Claude 3.5 Sonnet consistently outperformed GPT-4o. The authors conclude that Claude 3.5 Sonnet demonstrates higher reliability for AIS detection and localization than GPT-4o, though both models exhibit limitations that underscore the need for further refinement before routine clinical use. Table 1 and Supplementary Table 1 present examples of LLM applications in stroke detection and imaging interpretation. Data extraction One of the big advantages of NLP and specifically LLMs is extracting structured data from free-text. Critical data points are frequently buried in notes or reports, making them difficult to search and analyze. By automatically converting free-text information into structured formats—such as standardized fields for demographics, imaging results, or procedural timestamps—clinicians can quickly access key insights, identify patterns, and integrate these data into stroke registries or prediction models. 43 Lehnen et al. conducted a retrospective analysis of 130 free-text neuroradiology reports (100 internal and 30 external) from mechanical thrombectomy procedures, comparing GPT-4 and GPT-3.5 on their ability to automatically extract standardized procedural details. Their results showed that GPT-4 correctly retrieved 94% of data points (e.g., occlusion site, times, and materials used), substantially outperforming GPT-3.5 (64%) in both the main and external datasets. 44 Notably, prompt refinements further improved GPT-4’s performance on challenging fields like “last thrombectomy maneuver time,” underscoring the importance of prompt design. Similarly, Meddeb et al. evaluated three locally deployed open-source LLMs—Mistral’s Mixtral, 45 Alibaba’s Qwen, 46 and BioMistral fine-tuned model 47 —on more than 1000 mechanical thrombectomy reports, testing their ability to extract 15 clinical data fields (e.g., National Institutes of Health Stroke Scale [NIHSS] scores, occluded vessels, medication details). 48 Mixtral demonstrated particularly high precision (up to 0.99 for certain time metrics), while Qwen and BioMistral showed variable but still useful performance across different data fields. In addition, a human-in-the-loop (HITL) approach reduced the time required for final data labeling by over 65%. This underscores how partial automation can significantly expedite documentation without sacrificing accuracy. The results highlight the scalability of open-source LLMs for large-volume data extraction Goh et al. examined the feasibility of using a locally deployed LLM (Meta’s Llama 3) 28 to automate stroke audit data extraction from free-text discharge summaries. By comparing the model’s outputs to a human-maintained statewide stroke registry, they found a 93.8% accuracy rate across multiple data fields (e.g., wake-up stroke status, past medical history, pre-stroke medications), highlighting the potential of LLMs to streamline and improve consistency in audit processes. 6 Table 2 and Supplementary Table 2 present examples of LLM applications in stroke data extraction. View this table: View inline View popup Download powerpoint Table 2: Examples of LLM applications in stroke data extraction Outcome Prediction Predicting stroke outcomes is an evolving challenge, as conventional scoring systems like the MT-DRAGON score often fail to incorporate real-time, dynamic patient data. 49 AI has emerged as a powerful alternative by integrating multimodal inputs, including clinical parameters, imaging findings, and time-dependent variables. In a pilot study, Pedro et al. assessed ChatGPT’s performance in predicting post-thrombectomy functional outcomes, finding that it outperformed the MT-DRAGON score in forecasting modified Rankin Scale (mRS) scores at three months. Notably, AI predictions were more accurate for patients with shorter onset-to-door delays and better reperfusion status, suggesting that LLMs could refine post-stroke prognosis stratification. 50 Phillips et al. introduce “HELMET,” a hybrid machine learning framework that combines a fine-tuned LLMs with structured electronic health record (EHR) variables to predict malignant cerebral edema in large middle cerebral artery (MCA) stroke. 51 They trained and validated two models (HELMET-8 and HELMET-24) on a retrospective cohort of 623 patients from two Massachusetts hospitals (yielding over 15,000 patient-hour observations) and then externally tested on 60 patients at a separate safety-net hospital (over 3,700 observations). By continuously evaluating the degree of midline shift (MLS)—a critical indicator of severe edema—across 8-hour and 24-hour horizons, HELMET outperforms simpler regression-based scores and generalizes effectively to an external validation cohort. The authors aim to provide a real-time, dynamic risk stratification tool that can help clinicians monitor and intervene on malignant edema as it evolves, rather than relying on static early predictions alone. Table 3 and Supplementary Table 3 present examples of LLM applications in stroke outcome prediction. View this table: View inline View popup Download powerpoint Table 3: Examples of LLM applications in stroke outcome prediction Rehabilitation and Long-Term Management Stroke rehabilitation is a highly individualized process, yet AI-driven models may optimize therapy by predicting recovery potential and tailoring interventions. Zhang et al. explored ChatGPT-4’s ability to generate rehabilitation prescriptions and classify stroke patients according to ICF (International Classification of Functioning, Disability, and Health) codes, finding that it provided comprehensive, rational therapy plans. 52 While minor classification errors were noted, AI’s ability to generate structured rehabilitation strategies in seconds offers an opportunity to personalize post-stroke care. LLMs may also help with cognitive and language recovery. Cong et al. investigated AI’s application in aphasia assessment, demonstrating that pre-trained models could detect language deficits in stroke patients and even subtype aphasia based on linguistic patterns. Such models could augment traditional speech therapy, enabling automated conversational training tailored to individual deficits. 53 Table 4 and Supplementary Table 4 present examples of LLM applications in rehabilitation and long-term management. View this table: View inline View popup Download powerpoint Table 4: Examples of LLM applications in stroke rehabilitation Patient Education LLMs demonstrate potential in stroke-related patient education by explaining ischemic stroke pathophysiology, diagnostic work-up, and secondary prevention strategies in plain language. 54 , 55 In simulated patient interactions, these models promptly advised calling emergency services, and guided users through what to expect upon arrival at the emergency department, illustrating how real-time dialogue may enhance symptom recognition and prompt decision-making. 54 However, LLMs exhibit notable limitations, such as omitting major risk factors and therapeutic time windows, 55 and pose practical challenges for patients with dysarthria, dysphasia, aphasia, or alexia— barriers that must be addressed before clinical integration. 54 Table 5 and Supplementary Table 5 present examples of LLM applications in patient education. View this table: View inline View popup Download powerpoint Table 5: Examples of LLM applications in stroke patient education CHALLENGES AND LIMITATIONS The following cross-cutting weaknesses emerge across the 15 studies: Evidence quality Most performance claims rest on retrospective, single-center cohorts or curated case reports — settings that do not reflect the complexity and variability of real-world stroke care. These studies often lack diversity in stroke subtypes, presentation severity, and comorbidities, limiting their applicability across diverse clinical environments, particularly in EDs or telestroke settings. Reproducibility constraints Nine of the fifteen studies rely solely on proprietary APIs. 7 , 41 , 42 , 44 , 50 , 54 – 57 Neither code nor weights are publicly available, and full prompts are not always shared. This opacity prevents independent replication and stress-testing. Heterogeneous, incomparable benchmarks Studies report F1 score, AUC, accuracy, or mean absolute error on different tasks such as lesion localization, outcome prediction, or registry extraction. Without shared datasets or standard endpoints, cross-study comparisons are tenuous and risk misleading readers. Regulatory, legal, and privacy hurdles Most papers mention none of HIPAA, GDPR, FDA Software-as-a-Medical-Device guidance, or the EU AI Act. While LLMs may support rapid documentation or prediction, their integration into stroke workflows must comply with these legal frameworks. 58 Deployment pathways, cybersecurity standards, and liability frameworks remain uncharted. Equity and bias The studies generally did not stratify performance by sex, race, primary language, or insurance status —factors known to influence stroke outcomes. 59 LLMs trained on historical records may perpetuate existing treatment gaps unless bias audits and mitigation steps are built in. 60 , 61 Ethical and safety concerns Hallucinations, silent omissions, and over-confident outputs threaten patient safety. 6 , 42 , 50 Few studies report human-in-the-loop guardrails, real-time monitoring, or fallback protocols. 48 , 51 , 62 DISCUSSION In stroke care, “time is brain”. This core principle has driven developments such as in-hospital CT scanners and tele-stroke networks, all aimed at accelerating diagnosis and treatment. 63 We may now face a new technological inflection point: integrating LLMs into everyday practice to aid stroke detection, documentation, outcome prediction, and recovery planning. LLMs can review unstructured text from diverse sources, including real-time imaging logs and historical case reports, and may address persistent challenges in stroke care. 43 Early GPT-4 research suggests potential in automating tasks such as extracting procedural details from thrombectomy reports or synthesizing registry data for predictive analytics. 48 51 This could free stroke neurologists to focus on clinical judgment, patient communication, and team coordination. Preliminary evidence also hints that LLMs might forecast risks of malignant edema or long-term disability, enabling real-time predictions that adapt as a patient’s condition evolves. 51 Rigorous validation remains critical. “Hallucinations”—where a model offers a convincing but incorrect statement—underscore the danger of relying on AI without safeguards. 36 37 In a time-sensitive stroke activation, an erroneous conclusion could carry serious consequences. Human-in-the-loop systems that combine AI outputs with clinician expertise may mitigate this problem, with LLMs serving as advanced aids subject to ongoing real-world scrutiny. Equitable access to these tools also requires attention. Many smaller hospitals and safety-net facilities face resource constraints that hinder deployment of advanced AI software. 64 Models capable of running locally, securely, and at reasonable cost could help close this gap. Data privacy, liability concerns, and regulatory oversight must keep pace with rapid innovations. 65 Our review mapped some key challenges in the literature. Current evidence is narrow and fragile: most studies are retrospective, single-center, and occasionally exposed to data the models may have seen during pre-training. Proprietary APIs and undisclosed prompts block replication and external scrutiny. Metrics are reported on heterogeneous tasks, hampering comparison and masking over-fitting. Hallucinations and silent omissions remain an unresolved safety risk, especially where time-critical decisions are involved. Finally, future works should audit performance across sex, language, or socioeconomic features. Limitations of this review. Our search was confined to PubMed and Google Scholar and limited to English-language papers available up to 30 January 2025; relevant work in other languages may have been missed. As a narrative review, we did not apply formal risk-of-bias tools or meta-analytic methods, and performance numbers were taken as reported. Rapid advances in both model capability and regulation mean that findings may date quickly. Transformative shifts in stroke care typically arise from technologies that integrate into existing workflows and reduce morbidity and mortality. 66 LLMs, if refined through transparent validation, may offer a new complement to acute triage and procedure documentation. 67 , 68 AI-based language models may change how we interpret clinical data and guide decisions. In conclusion, LLMs are promising research instruments for mining stroke documentation, but the jump to bedside use demands multisite validation, open benchmarks, bias audits, and robust human oversight. Until those conditions are met, these models should augment—not replace— expert clinical judgment in time-sensitive stroke care. Author Contributions Shelly Soffer : Conceptualization, Methodology, Validation, Formal analysis, Investigation, Resources, Data Curation, Writing - Original Draft, Writing - Review & Editing, Visualization, Supervision, Project administration. Eyal Klang : Conceptualization, Methodology, Software, Validation, Formal analysis, Investigation, Resources, Data Curation, Writing - Review & Editing, Supervision, Project administration. Aya Mudrik, Orly Efros, Mahmud Omar, Girish N Nadkarni: Writing - Review & Editing, Validation, Supervision, Project administration. Competing interests The authors declare no competing interests. Data availability statement The datasets generated during and/or analyzed during the current study are available from the corresponding author on reasonable request. Funding sources This research did not receive any specific grant from funding agencies in the public, commercial, or not-for-profit sectors. References 1. ↵ Adeoye , O. et al. Recommendations for the Establishment of Stroke Systems of Care: A 2019 Update . Stroke 50 , ( 2019 ). 2. ↵ Wang , Y. et al. Clinical information extraction applications: A literature review . J Biomed Inform 77 , 34 – 49 ( 2018 ). OpenUrl CrossRef PubMed 3. ↵ Yang , X. et al. A large language model for electronic health records . NPJ Digit Med 5 , 194 ( 2022 ). OpenUrl PubMed 4. ↵ Meng , X. et al. The application of large language models in medicine: A scoping review . iScience 27 , 109713 ( 2024 ). OpenUrl PubMed 5. ↵ Song , X. et al. Stroke Diagnosis and Prediction Tool Using ChatGLM: Development and Validation Study . J Med Internet Res 27 , e67010 ( 2025 ). OpenUrl PubMed 6. ↵ Goh , R. et al. Large language models can effectively extract stroke and reperfusion audit data from medical free-text discharge summaries . Journal of Clinical Neuroscience 129 , 110847 ( 2024 ). OpenUrl PubMed 7. ↵ Koyun , M. & Taskent , I . Evaluation of Advanced Artificial Intelligence Algorithms’ Diagnostic Efficacy in Acute Ischemic Stroke: A Comparative Analysis of ChatGPT-4o and Claude 3.5 Sonnet Models . J Clin Med 14 , 571 ( 2025 ). OpenUrl PubMed 8. ↵ Neo , J. R. E. , Ser , J. S. & Tay , S. S . Use of large language model-based chatbots in managing the rehabilitation concerns and education needs of outpatient stroke survivors and caregivers . Front Digit Health 6 , 1395501 ( 2024 ). OpenUrl PubMed 9. ↵ Mudrik , A. et al. Exploring the role of Large Language Models in haematology: A focused review of applications, benefits and limitations . Br J Haematol ( 2024 ) doi: 10.1111/bjh.19738 . OpenUrl CrossRef 10. ↵ Marinho , V. et al. Impaired decision-making and time perception in individuals with stroke: Behavioral and neural correlates . Rev Neurol (Paris) 175 , 367 – 376 ( 2019 ). OpenUrl PubMed 11. ↵ Amisha , Malik , P. , Pathania , M. & Rathaur , V. K. Overview of artificial intelligence in medicine . J Family Med Prim Care 8 , 2328 – 2331 ( 2019 ). OpenUrl CrossRef PubMed 12. ↵ Koska , İ. Ö. & Selver , A. Artificial Intelligence in Stroke Imaging: A Comprehensive Review . Eurasian J Med 55 , 91 – 97 ( 2023 ). OpenUrl PubMed 13. ↵ Dresser , L. P. & Kohn , M. A. Artificial Intelligence and the Evaluation and Treatment of Stroke . Dela J Public Health 9 , 82 – 84 ( 2023 ). OpenUrl PubMed 14. ↵ Myllyaho , L. , Raatikainen , M. , Männistö , T. , Mikkonen , T. & Nurminen , J. K . Systematic literature review of validation methods for AI systems . Journal of Systems and Software 181 , 111050 ( 2021 ). OpenUrl CrossRef 15. ↵ Kufel , J. et al. What Is Machine Learning, Artificial Neural Networks and Deep Learning? -Examples of Practical Applications in Medicine. Diagnostics (Basel) 13 , ( 2023 ). 16. ↵ Sarker , I. H. Machine Learning: Algorithms, Real-World Applications and Research Directions . SN Comput Sci 2 , 160 ( 2021 ). OpenUrl PubMed 17. ↵ Khalifa , M. & Albadawy , M . Artificial Intelligence for Clinical Prediction: Exploring Key Domains and Essential Functions . Computer Methods and Programs in Biomedicine Update 5 , 100148 ( 2024 ). OpenUrl 18. ↵ Zou , J. , Han , Y. & So , S.-S . Overview of artificial neural networks . Methods Mol Biol 458 , 15 – 23 ( 2008 ). OpenUrl PubMed 19. ↵ Han , S.-H. , Kim , K. W. , Kim , S. & Youn , Y. C . Artificial Neural Network: Understanding the Basic Concepts without Mathematics . Dement Neurocogn Disord 17 , 83 – 89 ( 2018 ). OpenUrl PubMed 20. ↵ Soffer , S. et al. Convolutional Neural Networks for Radiologic Images: A Radiologist’s Guide . Radiology 290 , 590 – 606 ( 2019 ). OpenUrl CrossRef PubMed 21. ↵ Sorin , V. , Barash , Y. , Konen , E. & Klang , E . Deep Learning for Natural Language Processing in Radiology-Fundamentals and a Systematic Review . J Am Coll Radiol 17 , 639 – 648 ( 2020 ). OpenUrl CrossRef PubMed 22. ↵ Sorin , V. , Barash , Y. , Konen , E. & Klang , E . Deep-learning natural language processing for oncological applications . Lancet Oncol 21 , 1553 – 1556 ( 2020 ). OpenUrl CrossRef PubMed 23. ↵ Hossain , E. et al. Natural Language Processing in Electronic Health Records in relation to healthcare decision-making: A systematic review . Comput Biol Med 155 , 106649 ( 2023 ). OpenUrl CrossRef PubMed 24. ↵ Vaswani A , Shazeer N , et al. Attention Is All You Need . arXiv . 2017 . doi: 10.48550/arXiv.1706.03762 . OpenUrl CrossRef 25. ↵ OpenAI . Hello GPT-4o . 13 May 2024, https://openai.com/index/hello-gpt-4o/ . 26. ↵ OpenAI . ‘Introducing OpenAI o1.’ https://openai.com/o1/ 27. ↵ Anthropic . Introducing Claude 3.5 Sonnet . 20 June 2024, https://www.anthropic.com/news/claude-3-5-sonnet . 28. ↵ Meta . LLaMA 3: Model Cards and Prompt Formats . 2024 , https://www.llama.com/docs/model-cards-and-prompt-formats/llama3_3/ . 29. ↵ Daya Guo , Dejian Yang , Haowei Zhang , Junxiao Song , Ruoyu Zhang , Runxin Xu , Qihao Zhu , Shirong Ma , Peiyi Wang , Xiao Bi , et al. 2025 . Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning . Retrieved from https://arxiv.org/abs/2501.12948 30. ↵ Anisuzzaman , D. M. , Malins , J. G. , Friedman , P. A. & Attia , Z. I . Fine-Tuning Large Language Models for Specialized Use Cases . Mayo Clinic Proceedings: Digital Health 3 , 100184 ( 2025 ). 31. ↵ Liu , Lingjiao , Wenhao Zhou , and Tongshuang Zhang . Large Language Models are Easily Distracted: How Attention Fluctuates with Prompts . arXiv preprint arXiv: 2408.13296 , 2024 . 32. ↵ Patel , D. et al. Evaluating prompt engineering on GPT-3.5’s performance in USMLE-style medical calculations and clinical scenarios generated by GPT-4 . Sci Rep 14 , 17341 ( 2024 ). OpenUrl PubMed 33. ↵ Meskó , B . Prompt Engineering as an Important Emerging Skill for Medical Professionals: Tutorial . J Med Internet Res 25 , e50638 ( 2023 ). OpenUrl CrossRef PubMed 34. ↵ Fink , A. , Rau , A. , Kotter , E. , Bamberg , F. & Russe , M. F . [Optimized interaction with Large Language Modelslll: A practical guide to Prompt Engineering and Retrieval-Augmented Generation] . Radiologie (Heidelberg, Germany) ( 2025 ) doi: 10.1007/s00117-025-01416-2 . OpenUrl CrossRef 35. ↵ Zaghir , J. et al. Prompt Engineering Paradigms for Medical Applications: Scoping Review . J Med Internet Res 26 , e60501 ( 2024 ). OpenUrl CrossRef PubMed 36. ↵ Xiao , Y. & Wang , W. Y . On hallucination and predictive uncertainty in conditional language generation . In Proc. 16th Conference of the European Chapter of the Association for Computational Linguistics 2734–2744 ( Association for Computational Linguistics , 2021 )., https://ai.nejm.org/doi/full/10.1056/AIdbp2300040 . 37. ↵ Farquhar , S. , Kossen , J. , Kuhn , L. & Gal , Y . Detecting hallucinations in large language models using semantic entropy . Nature 630 , 625 – 630 ( 2024 ). OpenUrl CrossRef PubMed 38. ↵ Green , B. N. , Johnson , C. D. & Adams , A . Writing narrative literature reviews for peer-reviewed journals: secrets of the trade . J Chiropr Med 5 , 101 – 17 ( 2006 ). OpenUrl CrossRef PubMed 39. ↵ Ferrari , R . ( 2015 ). Writing narrative style literature reviews . Medical Writing , 24 ( 4 ), 230 – 235 . doi: 10.1179/2047480615Z.000000000329 . OpenUrl CrossRef 40. ↵ OpenAI . ( 2023 ). GPT-4 . OpenAI . https://openai.com/index/gpt-4/ . 41. ↵ Lee , J.-H. , Choi , E. , McDougal , R. & Lytton , W. W . GPT-4 Performance for Neurologic Localization . Neurol Clin Pract 14 , e200293 ( 2024 ). OpenUrl PubMed 42. ↵ Wang , X. et al. Performance of ChatGPT on prehospital acute ischemic stroke and large vessel occlusion (LVO) stroke screening . Digit Health 10 , 20552076241297130 ( 2024 ). 43. ↵ Huang , J. et al. A critical assessment of using ChatGPT for extracting structured data from clinical notes . NPJ Digit Med 7 , 106 ( 2024 ). 44. ↵ Lehnen , N. C. , et al. Data Extraction from Free-Text Reports on Mechanical Thrombectomy in Acute Ischemic Stroke Using ChatGPT: A Retrospective Analysis . Radiology 311 , e232741 ( 2024 ). OpenUrl CrossRef PubMed 45. ↵ Mistral AI . Mixtral of Experts . 8 Dec. 2023 , https://mistral.ai/news/mixtral-of-experts . 46. ↵ Alibaba Cloud . Qwen: Generative AI Solution . https://www.alibabacloud.com/en/solutions/generative-ai/qwen?_p_lc=1 . 47. ↵ Zhou , Junyang , et al. Qwen2: Scaling Instruction Tuning for Open-Domain Alignment . arXiv preprint arXiv : 2402.10373 , 2024 . https://arxiv.org/abs/2402.10373 . 48. ↵ Meddeb , A. et al. Evaluating local open-source large language models for data extraction from unstructured reports on mechanical thrombectomy in patients with ischemic stroke . J Neurointerv Surg ( 2025 ) doi: 10.1136/jnis-2024-022078 . OpenUrl Abstract / FREE Full Text 49. ↵ Lesenne , A. et al. Prediction of Functional Outcome After Acute Ischemic Stroke: Comparison of the CT-DRAGON Score and a Reduced Features Set . Front Neurol 11 , 718 ( 2020 ). 50. ↵ Pedro , T. et al. Exploring the use of ChatGPT in predicting anterior circulation stroke functional outcomes after mechanical thrombectomy: a pilot study . J Neurointerv Surg 17 , 261 – 265 ( 2025 ). OpenUrl Abstract / FREE Full Text 51. ↵ Phillips , E. et al. HELMET: A Hybrid Machine Learning Framework for Real-Time Prediction of Edema Trajectory in Large Middle Cerebral Artery Stroke . Preprint at doi: 10.1101/2024.11.13.24317229 ( 2024 ). OpenUrl Abstract / FREE Full Text 52. ↵ Zhang , L. , Tashiro , S. , Mukaino , M. & Yamada , S . Use of artificial intelligence large language models as a clinical tool in rehabilitation medicine: a comparative test case . J Rehabil Med 55 , jrm13373 ( 2023 ). OpenUrl CrossRef PubMed 53. ↵ Cong , Y. , LaCroix , A. N. & Lee , J . Clinical efficacy of pre-trained large language models through the lens of aphasia . Sci Rep 14 , 15573 ( 2024 ). OpenUrl PubMed 54. ↵ Lam , W. & Au , S. C . Stroke care in the ChatGPT era: Potential use in early symptom recognition . Journal of Acute Disease 12 , 129 ( 2023 ). OpenUrl 55. ↵ Vora , N. N. & Doshi , P. K . Use of ChatGPT in Creating Awareness about Ischemic Stroke . Indian J Community Med 48 , 633 – 635 ( 2023 ). OpenUrl PubMed 56. Wang , M. et al. Precision Structuring of Free-Text Surgical Record for Enhanced Stroke Management: A Comparative Evaluation of Large Language Models . J Multidiscip Healthc 17 , 5163 – 5175 ( 2024 ). OpenUrl PubMed 57. ↵ Kuzan , B. N. , Meşe , İ. , Yaşar , S. & Kuzan , T. Y. A retrospective evaluation of the potential of ChatGPT in the accurate diagnosis of acute stroke . Diagn Interv Radiol 31 , 187 – 195 ( 2025 ). OpenUrl PubMed 58. ↵ Ong , J. C. L. et al. Ethical and regulatory challenges of large language models in medicine . Lancet Digit Health 6 , e428 – e432 ( 2024 ). OpenUrl 59. ↵ Alawieh , A. , Zhao , J. & Feng , W . Factors affecting post-stroke motor recovery: Implications on neurotherapy after brain injury . Behavioural brain research 340 , 94 – 101 ( 2018 ). OpenUrl PubMed 60. ↵ Omar , M. et al. Sociodemographic biases in medical decision making by large language models . Nat Med ( 2025 ) doi: 10.1038/s41591-025-03626-6 . OpenUrl CrossRef PubMed 61. ↵ Omar , M. et al. Evaluating and addressing demographic disparities in medical large language models: a systematic review . Int J Equity Health 24 , 57 ( 2025 ). OpenUrl PubMed 62. ↵ Kim , J. et al. PhenoFlow: A Human-LLM Driven Visual Analytics System for Exploring Large and Complex Stroke Datasets . IEEE Trans Vis Comput Graph 31 , 470 – 480 ( 2025 ). OpenUrl PubMed 63. ↵ Bakka , A. G. et al. Breaking Barriers in Stroke Therapy: Recent Advances and Ongoing Challenges . Cureus 17 , e78288 ( 2025 ). OpenUrl 64. ↵ Rajkomar , A. , Hardt , M. , Howell , M. D. , Corrado , G. & Chin , M. H . Ensuring Fairness in Machine Learning to Advance Health Equity . Ann Intern Med 169 , 866 – 872 ( 2018 ). OpenUrl CrossRef PubMed 65. ↵ Mudrik , A. et al. Leveraging Large Language Models in Gynecologic Oncology: A Systematic Review of Current Applications and Challenges . Preprint at doi: 10.1101/2024.08.08.24311699 ( 2024 ). OpenUrl Abstract / FREE Full Text 66. ↵ Khosla , A. et al. An integrated machine learning approach to stroke prediction . in Proceedings of the 16th ACM SIGKDD international conference on Knowledge discovery and data mining 183 – 192 ( ACM , New York, NY, USA , 2010 ). doi: 10.1145/1835804.1835830 . OpenUrl CrossRef 67. ↵ Riedemann , L. , Labonne , M. & Gilbert , S . The path forward for large language models in medicine is open . NPJ Digit Med 7 , 339 ( 2024 ). OpenUrl CrossRef PubMed 68. ↵ Williams , C. Y. K. et al. Use of a Large Language Model to Assess Clinical Acuity of Adults in the Emergency Department . JAMA Netw Open 7 , e248895 ( 2024 ). OpenUrl View the discussion thread. Back to top Previous Next Posted July 01, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Large Language Models in Stroke Management: A Review of the Literature Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Large Language Models in Stroke Management: A Review of the Literature Shelly Soffer , Aya Mudrik , Orly Efros , Mahmud Omar , Girish N Nadkarni , Eyal Klang medRxiv 2025.06.28.25330477; doi: https://doi.org/10.1101/2025.06.28.25330477 Share This Article: Copy Citation Tools Large Language Models in Stroke Management: A Review of the Literature Shelly Soffer , Aya Mudrik , Orly Efros , Mahmud Omar , Girish N Nadkarni , Eyal Klang medRxiv 2025.06.28.25330477; doi: https://doi.org/10.1101/2025.06.28.25330477 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Neurology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (541) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a008783bc9f1fff4',t:'MTc3OTU4NTU0MA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.