Full text
39,650 characters
· extracted from
preprint-html
· click to expand
Evaluating GPT-4 as a Clinical Decision Support Tool in Ischemic Stroke Management | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Evaluating GPT-4 as a Clinical Decision Support Tool in Ischemic Stroke Management Amit Haim , Mark Katson , View ORCID Profile Michal Cohen-Shelly , Shlomi Peretz , View ORCID Profile Dvir Aran , View ORCID Profile Shahar Shelly doi: https://doi.org/10.1101/2024.01.18.24301409 Amit Haim 1 Department of Neurology, Rambam Medical Center , Haifa, Israel Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mark Katson 1 Department of Neurology, Rambam Medical Center , Haifa, Israel MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Michal Cohen-Shelly 2 Sagol AI Hub, ARC Innovation Center, Chaim Sheba Medical Center Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michal Cohen-Shelly Shlomi Peretz 3 Department of Neurology, Shamir Medical Center, Tzrifin, Sackler Faculty of Medicine, Tel Aviv University , Tel Aviv, Israel MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Dvir Aran 4 Faculty of Biology, Technion-Israel Institute of Technology , Haifa, Israel 5 The Taub Faculty of Computer Science, Technion-Israel Institute of Technology , Haifa, Israel PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Dvir Aran Shahar Shelly 1 Department of Neurology, Rambam Medical Center , Haifa, Israel 6 Rapaport Faculty of Medicine, Technion – Israel Institute of Technology , Haifa, 3525408, Israel 7 Department of Neurology, Mayo Clinic , Rochester, Minnesota MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Shahar Shelly For correspondence: shahar.shell{at}technion.ac.il Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Cerebrovascular diseases are the second most common cause of death worldwide and one of the major causes of disability burden. Advancements in artificial intelligence (AI) have the potential to revolutionize healthcare delivery, particularly in critical decision-making scenarios such as ischemic stroke management. This study evaluates the effectiveness of GPT-4 in providing clinical decision support for emergency room neurologists by comparing its recommendations with expert opinions and real-world treatment outcomes. A cohort of 100 consecutive patients with acute stroke symptoms was retrospectively reviewed. The data used for decision making included patients’ history, clinical evaluation, imaging studies results, and other relevant details. Each case was independently presented to GPT-4, which provided a scaled recommendation (1-7) regarding the appropriateness of treatment, the use of tissue plasminogen activator (tPA), and the need for endovascular thrombectomy (EVT). Additionally, GPT-4 estimated the 90-day mortality probability for each patient and elucidated its reasoning for each recommendation. The recommendations were then compared with those of a stroke specialist and actual treatment decision. The agreement of GPT-4’s recommendations with the expert opinion yielded an Area Under the Curve (AUC) of 0.85 [95% CI: 0.77-0.93], and with real-world treatment decisions, an AUC of 0.80 [0.69-0.91]. In terms of mortality prediction, out of 13 patients who died within 90 days, GPT-4 accurately identified 10 within its top 25 high-risk predictions (AUC = 0.89 [95% CI: 0.8077-0.9739]; HR: 6.98 [95% CI: 2.88-16.9]), surpassing supervised machine-learning models. This study demonstrates the potential of GPT-4 as a viable clinical decision support tool in the management of ischemic stroke. Its ability to provide explainable recommendations without requiring structured data input aligns well with the routine workflows of treating physicians. Future studies should focus on prospective validations and exploring the integration of such AI tools into clinical practice. Introduction The advent of GPT-4, launched by OpenAI in March 2023, marked a significant milestone in the evolution of artificial intelligence (AI) and its applications in various domains, including healthcare. GPT-4, a model under the umbrella of Generative Pretrained Transformers (GPT), exemplifies the advancement in large language model (LLM) technology. 1 , 2 The foundational architecture of this technology involves training on extensive datasets, enabling the model to function as a ‘few-shot learner’. This capability allows GPT-4 to adapt to new domains and continuously refine its performance through ongoing learning. 1 , 3 In the realm of clinical medicine, the potential applications of LLMs like GPT-4 are particularly intriguing. These models offer promise as supportive tools for healthcare professionals, aiding in the efficient summarization of patient data, assisting in decision-making processes, and potentially improving the accuracy and speed of medical interventions. 4 , 5 Recent research has underscored the capabilities of GPT-4 in complex medical tasks. Notably, the model has demonstrated proficiency in examinations akin to the United States Medical Licensing Examination (USMLE), achieving scores that meet or nearly meet the passing thresholds. 6 Additionally, in assessments modeled after neurology board exam questions, GPT-4 has shown a high accuracy rate, improving with repeated attempts. 7 , 8 The management of acute ischemic stroke (AIS) presents a critical and time-sensitive challenge in clinical settings. The approach to diagnosing and treating AIS requires a synthesis of information including patient symptoms, physical and neurological examinations, medical history, and imaging results. Despite the availability of established guidelines by the American Heart Association/American Stroke Association (AHA/ASA) for stroke management, 9 - 11 the pivotal role of the treating physician’s judgment remains. Variability in clinical presentations and the urgent need for decision-making underscore the potential value of AI-assisted tools in this context. Moreover, predicting early mortality in AIS is essential for guiding treatment decisions, optimizing resource allocation in healthcare settings, facilitating effective communication with patients and their families, supporting research and clinical trials, and contributing to quality improvement initiatives. In accordance, several traditional machine-learning models have been trained for this task in recent years. 12 - 14 Here, we leveraged patient data from the emergency department of a large referral hospital, focusing on individuals presenting with stroke symptoms, to evaluate the effectiveness of GPT-4 in delivering accurate clinical decisions for the treatment of AIS. We also assessed its proficiency in predicting 90-day mortality outcomes. We aimed to quantify the extent to which an advanced language model like GPT-4 can augment the clinical decision-making process, potentially contributing to improved patient outcomes in one of the most critical areas of emergency medicine. Results Patient demographics and clinical data We generated a cohort from 100 consecutive cases of patients presenting with acute stroke symptoms at the emergency department of Rambam Healthcare Campus. All cases underwent full clinical and radiological evaluation in the emergency setting for acute stroke and were fully evaluated by a neurologist ( Table 1 ; Figure 1A ). Revascularization treatment was administered to 78 of the patients: 36 were treated with tPA, 30 with EVT, and 12 received both. Within this cohort, 13 patients died within 90 days and 21 in total. Overall, 17 cases were classified as ‘complex’ when not fitting exact treatment guidelines. 9 The data for each case encompassed demographics, NIHSS 15 scores, timing of arrival to brain CT, onset of symptoms, and details from textual brain imaging results and risk factors that were available as medical history at the time of admission to the ER ( Supplementary Table 1 ). View this table: View inline View popup Download powerpoint Table 1. Study cohort clinical information and demographics Download figure Open in new tab Figure 1. Study Design and GPT-4 Performance Evaluation. A . Illustration of the study design involving 100 consecutive stroke patients who underwent a comprehensive stroke workup, including perfusion, angiography, and non-contrast brain CT upon arrival at the Emergency Room. Clinical information, demographics, comorbidities, and CT perfusion results were recorded. The textual reports from these investigations were entered into the GPT-4 API, which was instructed to provide scores indicating whether to treat the patient, whether to administer tPA, whether to pursue EVT, and an estimate of 90-days mortality (Created with BioRender.com ). B . Boxplots presenting average scores of GPT-4 assessments for decision to treat (y-axis). The comparison is made against real-world decisions and expert assessments of each case (TRUE – to treat the patient, FALSE – to not treat). C . ROC curves and AUC scores of GPT-4 average scores for decision to treat, compared to real-world decisions and expert assessments. A stroke specialist, blinded to the outcomes, retrospectively reviewed each case. In 82 of the cases, the expert’s decisions aligned with the actual treatments administered. Of note, the expert recommended against treating 11 patients who received treatment and suggested treatment for 7 who did not receive any. Concerning specific treatments, full agreement was observed in 61 cases, although the expert more frequently recommended combining tPA and EVT than what was observed in practice (Cohen’s Kappa = 0.51, signifying moderate agreement). GPT-4 clinical decisions Independently, each case was assessed with GPT-4, generating a treatment recommendation scale from 1 (intervention not recommended) to 7 (highly recommended) ( Figure 1A ; Supplementary Table 2 ). To account for the variability in GPT-4 responses, each case was assessed 5 times. Cohen’s Kappa for treatment scores across runs ranged from 0.56 to 0.73. As expected, the pre-defined ‘complex’ cases demonstrated significantly greater variance between runs (p-value = 0.02). Comparing GPT-4’s treatment scale to both the expert’s decision and the actual treatment revealed that the average scores from GPT-4 for patients who were treated were, on average, 1.9 points higher than those not treated (p-value < 0.001), and 2.1-point difference in comparison to the expert decision (p-value < 0.001; Figure 1B ). The average scores provided an area under the ROC curve (AUC-ROC) of 0.80 [95% CI: 0.69-0.91] compared to real-world decisions, and 0.85 [95% CI: 0.77-0.93] compared to the expert decision ( Figure 1C ). These average scores for AUCs were higher than those of each independent run ( Supplementary Figure 1 ). Additionally, removing the clinical presentation narrative from GPT-4’s analysis resulted in a drop in AUC to 0.70 with real-world decisions and 0.72 with the expert ( Supplementary Figure 1 ), highlighting the importance of unstructured narrative data in treatment decision-making. Similarly, setting the temperature of GPT-4 to 0 resulted in AUC of 0.70 and 0.72 with the real-world and the expert, respectively, suggesting the need to allow GPT-4 more creativity to obtain better decisions. Download figure Open in new tab Supplementary Figure 1. GPT-4 Assessments Performance. Area under the curve (AUC) for GPT-4 decision to treatment scores of each of the individual submissions (1-5) and the average. Each individual submission is lower than the average. In addition, we submitted the cases without the clinical presentation narrative, which yielded lower AUC (no narrative). Similarly, lower AUC was observed when cases were submitted with temperature=0. Using a score threshold of 4, we observed 22 disagreements between GPT-4 and real-world treatment and 20 disagreements with the expert decision. Notably, a significant proportion of these disagreements coincided with cases where the expert and real-world decisions diverged, with 18 out of 30 such cases showing this dual disagreement. Moreover, complex cases were more prone to discrepancies, as 7 disagreements with real-world and 5 with the expert were noted among the 17 complex cases. The specialist examined the explanatory text produced by GPT-4 for all discrepancies between the model and their blinded assessments, evaluating whether they agreed that the explanatory text, as part of the original model output, was logical and could be deemed good practice. Of the 20 instances where disagreements occurred, in three cases the expert, after having carefully considered GPT-4’s detailed explanations, conceded that GPT-4’s assessment was preferable to their original decision. In additional two cases the expert acknowledged that GPT-4’s suggested approach was indeed acceptable and aligned with viable treatment options. In instances where the expert disagreed with GPT-4’s reasoning, the disagreements primarily revolved around three key issues. Firstly, GPT-4 inaccurately associated abnormal angiographic findings with clinical presentations. An illustrative case is that of a patient with stenosis of the right-sided middle cerebral artery (MCA) who presented with right hemiparesis (case 94). Despite these two elements potentially being anatomically unrelated, GPT-4 linked them erroneously. The second notable issue pertained to ethical considerations, particularly in a case involving a patient with active laryngeal cancer and cognitive decline. According to guidelines, the patient was deemed eligible for treatment, but the expert’s decision was to not proceed with treatment as life expectancy was short and he was palliative (case 14). Thirdly, discrepancies arose in deviations from guidelines, particularly in cases of distal thrombectomies. For instance, in the case of an over 90 year-old patient with M2 obstruction (considered distal thrombus), GPT-4 recommended against treatment, which is the established guidelines, however, the expert call was to proceed with thrombectomy due to high NIHSS score and good results in such cases in the past from personal experience (case 54). In assessing GPT-4’s ability to choose the best treatment option, it showed near-perfect agreement with real-world decisions in recommending EVT: GPT-4 suggested EVT for all patients (42 of the 42) treated with EVT (average score >4). The expert suggested EVT for 55 patients, of which 50 were also recommended EVT by GPT-4, corresponding to an AUC of 0.94 [95% CI: 0.89-0.98] with real-world decisions and 0.95 [95% CI: 0.90-0.99] with the expert ( Figure 2A ). For tPA treatment, GPT-4 recommended it for 38 of the 48 patients who received it, showing a closer agreement with the expert. Of the 41 patients recommended for tPA by the expert, GPT-4 agreed on 35, corresponding to an AUC of 0.77 [95% CI: 0.68-0.86] with real-world decisions and 0.82 [95% CI: 0.73-0.90] with the expert ( Figure 2B ). Download figure Open in new tab Figure 2. GPT-4 treatment type scores. A . Boxplots depict GPT-4 treatment type scores, with the Y-axis representing probability score (1-7 scale). Each treatment category is color-coded: green for no intervention, orange for tPA (Tissue Plasminogen Activator), purple for Endovascular treatment (EVT), and pink for tPA and EVT. A . GPT-4 scores for EVT, stratified by real-world decisions and expert assessments. B . GPT-4 scores for tPA, stratified by real-world decisions and expert assessments. Mortality risk We further evaluated the ability of GPT-4 to predict 90-day mortality. The model estimated an average mortality risk of 55.1% for patients who died within 90 days, compared to 31.5% for survivors (p-value < 0.001), yielding an AUC of 0.89 [95% CI: 0.8077-0.9739] ( Figure 3A ). To contextualize these results, we compared GPT-4’s performance with that of two recent machine-learning models specifically trained for 90-day mortality prediction. In our cohort, the PRACTICE model 13 achieved an AUC of 0.70, significantly worse than the GPT-4 predictions (log-rank p-value = 0.02), while the PREMISE model 14 reached an AUC of 0.77 (p-value = 0.07) ( Figure 3A ). These comparisons underscore GPT-4’s remarkable accuracy in mortality risk assessment, outperforming specialized, trained predictive models. Download figure Open in new tab Figure 3. GPT-4 Mortality Predictions. A . ROC curve for 90-day mortality estimations by GPT-4 (red), PRACTICE (green), and PREMISE (blue). B . Kaplan-Meier plot stratifying individuals into low and high-risk categories for mortality based on GPT-4’s 90-day mortality estimations. For identifying high-risk patients, we set a threshold at the top 25% of the cohort, which corresponded to a predicted mortality risk cut-off of 41%. Within this high-risk group, 10 patients passed away within 90 days of admission, and an additional 3 within the subsequent year ( Figure 3B ). Conversely, among the remaining 75 patients categorized as lower risk, only 3 deaths occurred within the 90-day period, and 6 in total during the first year. The calculated Hazard Ratio was 6.98 [95% CI 2.88-16.9; p-value <0.001], reinforcing the model’s capability in stratifying patients based on their mortality risk effectively. Discussion This study introduces a pioneering application of a LLM predictive model, specifically GPT 4, to address acute ischemic stroke, the second most common cause of mortality and major cause of disability. 17 - 19 The urgency of stroke care is particularly magnified in regions where access to specialized stroke units or even qualified physicians is limited, especially in countryside areas. 20 , 21 The time-sensitive nature of stroke prognosis underscores the critical need for swift and accurate decision-making in these under-resourced healthcare facilities. The utility of GPT-4 in clinical practice is highlighted by its ability to operate seamlessly within existing treatment routines. 4 The model relies solely on routine chart information available in emergency settings, making it particularly valuable in regions with limited access to neurology experts or areas with high patient volume and admission rates necessitate quick triage. This accessibility could democratize high-level medical consultation, extending expert-level decision-making to under-resourced healthcare facilities. In our study, we also utilized GPT-4 to assess the predictive capacity for 90-day mortality in patients undergoing endovascular treatment for acute ischemic stroke. The GPT-4 model, utilizing a diverse range of clinical and imaging variables, demonstrated high accuracy in estimating mortality risk. Notably, the variables considered by GPT-4 encompass a wider spectrum, including clinical and imaging factors, offering a more comprehensive approach compared to existing models such as older scores such as HIAT and HIAT2, and newer scores such as PREMISE and PRACTICE. 13 , 14 , 22 , 23 Traditionally, healthcare predictive models rely heavily on collecting vast amounts of structured data and training specific machine learning algorithms. Contrarily, GPT-4 breaks this mold by providing comprehensive treatment recommendations and mortality risk based on narrative text, which is complicated to model in traditional machine-learning models. Our analyses highlighted the significance of unstructured data, as evidenced by the drop in prediction accuracy when the narrative clinical presentation was excluded. While traditional machine learning models struggle to process and interpret unstructured text, GPT-4 does so with apparent ease, showcasing its capability to handle complex medical data in a way that is more aligned with the natural flow of clinical information. A crucial aspect of GPT-4’s application in healthcare is its explainability. Deep learning models often face challenges in providing clear reasoning for their decisions, a significant barrier in clinical practice where understanding the rationale behind a recommendation is crucial. Our analysis of the decision rationale provided by GPT-4 has further demonstrated its ability to effectively highlight key clinical considerations. The expert’s review confirmed that GPT-4’s explanations were not only correct but also offered valuable insights, thereby underscoring the model’s potential as an instrumental aid in clinical decision-making. This level of explainability enhances trust in AI-assisted decision-making and could pave the way for broader acceptance and integration of AI tools in medical settings. Despite its promising results, our study has several limitations. We must acknowledge certain challenges in applying GPT-4, especially regarding its ability to assess ethical issues. The model may face difficulties in addressing the nuanced and complex ethical considerations intrinsic to medical decision-making. This limitation emphasizes the necessity for cautious and supplementary human oversight when deploying AI tools like GPT-4 in sensitive healthcare contexts. The occurrence of ‘hallucinations’ or erroneous outputs is another concern, although we demonstrated that running multiple assessments can mitigate this risk. Future research should focus on refining these methods to further reduce inaccuracies. Another consideration is the generalizability of these findings. Our study was conducted in a single center with a specific patient population. Further studies across diverse settings and larger populations are necessary to validate the efficacy and applicability of GPT-4 in various clinical environments. In conclusion, our study introduces a groundbreaking approach to clinical decision support in stroke management using GPT-4. This model has shown the potential to process narrative text, provide explainable recommendations and enhance medical decision-making. As we continue to explore and refine this technology, it holds the promise of transforming patient care and improving outcomes in one of the most critical areas of medicine. Methods Ethical approval Institutional review board/ethics committee approved this retrospective study in accidence of guidelines. Cohort Selection This retrospective study comprised 100 consecutive cases from the emergency department of Rambam Healthcare Campus. All patients treated between January 2022 and April 2023 received a confirmed diagnosis of acute ischemic stroke. The inclusion criteria encompassed patients older than 18, an NIHSS 15 score of 5 or higher(with the exception of patients 93 who received tPA off site), and less than 5 hours from symptom onset to undergoing a non-contrast CT of the brain. All included patients underwent non-contrast brain CT, CT angiography, and CT perfusion while in the ER. This cohort was specifically chosen for its alignment with AHA guidelines for acute stroke management 9 , making each patient a potential candidate for both tPA and EVT treatment. Seventeen patients, not meeting these criteria, were categorized as “complex” cases which the clinical scenario warranted extra consideration of off-guideline treatment options, and there was a need to assess the individual patient’s unique characteristics, medical history, and condition. For every patient, comprehensive medical records from their ER arrival, including imaging results, were collected, and translated from Hebrew to English. Exclusion criteria were patients with incomplete clinical data or where stroke was not the final diagnosis. Clinical data for each patient included demographics, medical history, chief complaints, symptom onset time, physical and neurological examinations, NIHSS score, imaging results (including ASPECTS 16 when available), treatment received, and mortality data. An experienced stoke specialist, blind to the outcomes, reviewed the cases and made treatment decisions among no treatment, tPA, EVT, or a combination of tPA and EVT. All data was deidentified, removing identifiers, names and dates. Analysis Pipeline The analysis utilized the OpenAI API ‘create chat completion’ method with the model gpt-4-1106-preview. Default parameters were set (temperature = 1, top_p = 1, n = 1), and submissions were made using the R wrapper library ‘openai’. The full prompt given to GPT-4 was as follows: “Imagine you are a board-certified neurologist in the emergency room. You are receiving a clinical case. Describe the best neurological approach leading to the best neurological outcome, and the lowest chance of mortality. Base your decision on the current guidelines and reason your decision. Note that in some cases patients should not be treated although the best treatment option due to the patient fragility . Here is your case: <> First, provide the full reasoning. Next, based on your response, answer the questions below. Only return a number, no additional reasoning. Provide the results in a structured format as following: [A, B,C,D], where A is answer for q1, B for q2, C for q3 and D for q4. For example, an answer could be [4,3,2,40] . Any intervention (tPA or EVT)? Answer with scale 1 to 7, where 1 is intervention not recommended and 7 is intervention is highly recommended . Thrombolytic therapy (tissue plasminogen activator; tPA)? Answer with scale 1 to 7, where 1 is tPA not recommended and 7 is tPA is the best option . Endovascular thrombectomy (EVT)? Answer with scale 1 to 7, where 1 is EVT not recommended and 7 is EVT is the best option . What is your estimation for 90-day mortality probability? Provide estimation even if there is not enough information. Use the scale 0 and 100 . ” To assess the reliability of GPT-4 responses, each case underwent five submissions, we well as an additional submission without the accompanying clinical presentation narrative. For every treatment decision, GPT-4 provided a narrative explanation. In 95% of cases, GPT-4 returned responses in the requested structure, which were automatically scraped with R. Unstructured responses were manually entered. For estimations provided as a range, the average was used. If GPT-4 provided a number with a greater symbol (e.g., >50), the number was recorded with an additional 5. In 0.8% of cases, GPT-4 did not return numeric responses for treatment decisions, and in 8.6% of responses, it did not provide a 90-day mortality estimate. Statistical Analysis GPT-4’s responses were scaled from 1 to 7 for treatment decisions and from 0 to 100 for 90-day mortality estimations. Averages were calculated across the five repeats. All statistical analyses were conducted using R version 4.3.2, employing base R functions, pROC 1.18.5, and survival 3.5.7. ROC curves were smoothed. Agreement between treatment decisions was measured using a linear weighted Cohen’s kappa coefficient, utilizing the psych 2.3.12 library. Data Availability All data that was used in this study is available as a supplementary table. Competing interest declaration There are no competing interests. DA reports consulting fees from Carelon Digital Platforms. Funding No funding was use. Acknowledgment and Author contributions Study concept and design: SS & DA Acquisition of data: SS, AM & DA Analysis and interpretation: DA, SS, MK & SP Draft of manuscript: DA & SS Critical revision of the manuscript: AM, MK, SP & MCS Reference 1. ↵ Sanderson K. GPT-4 is here: what scientists think . Nature . 2023 ; 615 ( 7954 ): 773 . OpenUrl CrossRef 2. ↵ Lee P , Bubeck S , Petro J. Benefits, Limits, and Risks of GPT-4 as an AI Chatbot for Medicine. Reply . N Engl J Med . 2023 ; 388 ( 25 ): 2400 . OpenUrl 3. ↵ Zack T , Lehman E , Suzgun M , et al. Assessing the potential of GPT-4 to perpetuate racial and gender biases in health care: a model evaluation study . Lancet Digit Health . 2024 ; 6 ( 1 ): e12 – e22 . OpenUrl 4. ↵ Kanjee Z , Crowe B , Rodman A. Accuracy of a Generative Artificial Intelligence Model in a Complex Diagnostic Challenge . JAMA . 2023 ; 330 ( 1 ): 78 – 80 . OpenUrl 5. ↵ Shea YF , Lee CMY , Ip WCT , Luk DWA , Wong SSW . Use of GPT-4 to Analyze Medical Records of Patients With Extensive Investigations and Delayed Diagnosis . JAMA Netw Open . 2023 ; 6 ( 8 ): e2325000 . OpenUrl 6. ↵ Brin D , Sorin V , Vaid A , et al. Comparing ChatGPT and GPT-4 performance in USMLE soft skill assessments . Sci Rep . 2023 ; 13 ( 1 ): 16492 . OpenUrl 7. ↵ Guillen-Grima F , Guillen-Aguinaga S , Guillen-Aguinaga L , et al. Evaluating the Efficacy of ChatGPT in Navigating the Spanish Medical Residency Entrance Examination (MIR): Promising Horizons for AI in Clinical Medicine . Clin Pract . 2023 ; 13 ( 6 ): 1460 – 1487 . OpenUrl 8. ↵ Guerra GA , Hofmann H , Sobhani S , et al. GPT-4 Artificial Intelligence Model Outperforms ChatGPT, Medical Students, and Neurosurgery Residents on Neurosurgery Written Board-Like Questions . World Neurosurg . 2023 ; 179 : e160 – e165 . OpenUrl 9. ↵ Kleindorfer DO , Towfighi A , Chaturvedi S , et al. 2021 Guideline for the Prevention of Stroke in Patients With Stroke and Transient Ischemic Attack: A Guideline From the American Heart Association/American Stroke Association . Stroke . 2021 ; 52 ( 7 ): e364 – e467 . OpenUrl CrossRef PubMed 10. Brown DL , Levine DA , Albright K , et al. Benefits and Risks of Dual Versus Single Antiplatelet Therapy for Secondary Stroke Prevention: A Systematic Review for the 2021 Guideline for the Prevention of Stroke in Patients With Stroke and Transient Ischemic Attack . Stroke . 2021 ; 52 ( 7 ): e468 – e479 . OpenUrl PubMed 11. ↵ Amin HP , Madsen TE , Bravata DM , et al. Diagnosis, Workup, Risk Reduction of Transient Ischemic Attack in the Emergency Department Setting: A Scientific Statement From the American Heart Association . Stroke . 2023 ; 54 ( 3 ): e109 – e121 . OpenUrl 12. ↵ Linfante I , Walker GR , Castonguay AC , et al. Predictors of Mortality in Acute Ischemic Stroke Intervention: Analysis of the North American Solitaire Acute Stroke Registry . Stroke . 2015 ; 46 ( 8 ): 2305 – 2308 . OpenUrl Abstract / FREE Full Text 13. ↵ Li H , Ye SS , Wu YL , et al. Predicting mortality in acute ischaemic stroke treated with mechanical thrombectomy: analysis of a multicentre prospective registry . BMJ Open . 2021 ; 11 ( 4 ): e043415 . OpenUrl Abstract / FREE Full Text 14. ↵ Gattringer T , Posekany A , Niederkorn K , et al. Predicting Early Mortality of Acute Ischemic Stroke . Stroke . 2019 ; 50 ( 2 ): 349 – 356 . OpenUrl PubMed 15. ↵ Kwah LK , Diong J. National Institutes of Health Stroke Scale (NIHSS) . J Physiother . 2014 ; 60 ( 1 ): 61 . OpenUrl PubMed 16. ↵ Pop NO , Tit DM , Diaconu CC , et al. The Alberta Stroke Program Early CT score (ASPECTS): A predictor of mortality in acute ischemic stroke . Exp Ther Med . 2021 ; 22 ( 6 ): 1371 . OpenUrl CrossRef PubMed 17. ↵ Lim GB . Global burden of cardiovascular disease . Nat Rev Cardiol . 2013 ; 10 ( 2 ): 59 . OpenUrl CrossRef PubMed 18. Feigin VL , Forouzanfar MH , Krishnamurthi R , et al. Global and regional burden of stroke during 1990-2010: findings from the Global Burden of Disease Study 2010 . Lancet . 2014 ; 383 ( 9913 ): 245 – 254 . OpenUrl CrossRef PubMed Web of Science 19. ↵ Collaborators GBDCoD . Global, regional, and national age-sex specific mortality for 264 causes of death, 1980-2016: a systematic analysis for the Global Burden of Disease Study 2016 . Lancet . 2017 ; 390 ( 10100 ): 1151 – 1210 . OpenUrl CrossRef PubMed 20. ↵ Saver JL , Fonarow GC , Smith EE , et al. Time to treatment with intravenous tissue plasminogen activator and outcome from acute ischemic stroke . JAMA . 2013 ; 309 ( 23 ): 2480 – 2488 . OpenUrl CrossRef PubMed Web of Science 21. ↵ Strbian D , Soinne L , Sairanen T , et al. Ultraearly thrombolysis in acute ischemic stroke is associated with better outcome and lower mortality . Stroke . 2010 ; 41 ( 4 ): 712 – 716 . OpenUrl Abstract / FREE Full Text 22. ↵ Ryu CW , Kim BM , Kim HG , et al. Optimizing Outcome Prediction Scores in Patients Undergoing Endovascular Thrombectomy for Large Vessel Occlusions Using Collateral Grade on Computed Tomography Angiography . Neurosurgery . 2019 ; 85 ( 3 ): 350 – 358 . OpenUrl 23. ↵ Hallevi H , Barreto AD , Liebeskind DS , et al. Identifying patients at high risk for poor outcome after intra-arterial therapy for acute ischemic stroke . Stroke . 2009 ; 40 ( 5 ): 1780 – 1785 . OpenUrl Abstract / FREE Full Text View the discussion thread. Back to top Previous Next Posted January 25, 2024. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Evaluating GPT-4 as a Clinical Decision Support Tool in Ischemic Stroke Management Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Evaluating GPT-4 as a Clinical Decision Support Tool in Ischemic Stroke Management Amit Haim , Mark Katson , Michal Cohen-Shelly , Shlomi Peretz , Dvir Aran , Shahar Shelly medRxiv 2024.01.18.24301409; doi: https://doi.org/10.1101/2024.01.18.24301409 Share This Article: Copy Citation Tools Evaluating GPT-4 as a Clinical Decision Support Tool in Ischemic Stroke Management Amit Haim , Mark Katson , Michal Cohen-Shelly , Shlomi Peretz , Dvir Aran , Shahar Shelly medRxiv 2024.01.18.24301409; doi: https://doi.org/10.1101/2024.01.18.24301409 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Neurology Subject Areas All Articles Addiction Medicine (573) Allergy and Immunology (865) Anesthesia (303) Cardiovascular Medicine (4456) Dentistry and Oral Medicine (445) Dermatology (383) Emergency Medicine (610) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1517) Epidemiology (15244) Forensic Medicine (30) Gastroenterology (1132) Genetic and Genomic Medicine (6619) Geriatric Medicine (669) Health Economics (1002) Health Informatics (4556) Health Policy (1372) Health Systems and Quality Improvement (1614) Hematology (543) HIV/AIDS (1270) Infectious Diseases (except HIV/AIDS) (15933) Intensive Care and Critical Care Medicine (1106) Medical Education (624) Medical Ethics (147) Nephrology (670) Neurology (6634) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1148) Occupational and Environmental Health (957) Oncology (3347) Ophthalmology (980) Orthopedics (369) Otolaryngology (421) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1696) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5463) Public and Global Health (9255) Radiology and Imaging (2210) Rehabilitation Medicine and Physical Therapy (1371) Respiratory Medicine (1197) Rheumatology (598) Sexual and Reproductive Health (716) Sports Medicine (532) Surgery (714) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a0308af9682d8e2e',t:'MTc4MDAwNTczOA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.