Full text
56,361 characters
· extracted from
preprint-html
· click to expand
irAE-GPT: Leveraging large language models to identify immune-related adverse events in electronic health records and clinical trial datasets | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search irAE-GPT: Leveraging large language models to identify immune-related adverse events in electronic health records and clinical trial datasets View ORCID Profile Cosmin A. Bejan , Michelle Wang , Sriram Venkateswaran , Ewa A. Bergmann , Laura Hiles , View ORCID Profile Yaomin Xu , G Scott Chandler , Sam Brondfield , Jordyn Silverstein , Francis Wright , Kimberly de Dios , Daniel Kim , View ORCID Profile Eric Mukherjee , Matthew S. Krantz , Lydia Yao , Douglas B. Johnson , Elizabeth J. Phillips , Justin M. Balko , Rajat Mohindra , Zoe Quandt doi: https://doi.org/10.1101/2025.03.05.25323445 Cosmin A. Bejan 1 Department of Biomedical Informatics, Vanderbilt University Medical Center , Nashville, USA Ph.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Cosmin A. Bejan For correspondence: adi.bejan{at}vanderbilt.edu Michelle Wang 2 Department of Bioengineering and Therapeutic Sciences, University of California San Francisco , San Francisco, USA 3 Bakar Computational Health Sciences Institute, University of California San Francisco , San Francisco, USA Pharm.D., Ph.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sriram Venkateswaran 4 F. Hoffmann-La Roche , Basel, Switzerland M.Sc. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ewa A. Bergmann 4 F. Hoffmann-La Roche , Basel, Switzerland Ph.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Laura Hiles 5 Roche Products Ltd , Welwyn Garden City, United Kingdom B.Pharm. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yaomin Xu 1 Department of Biomedical Informatics, Vanderbilt University Medical Center , Nashville, USA 6 Department of Biostatistics, Vanderbilt University Medical Center , Nashville, USA Ph.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Yaomin Xu G Scott Chandler 4 F. Hoffmann-La Roche , Basel, Switzerland M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sam Brondfield 7 Division of Hematology and Oncology, Department of Medicine, University of California San Francisco , San Francisco, USA 8 Helen Diller Family Comprehensive Cancer Center, University of California San Francisco , San Francisco, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jordyn Silverstein 9 Division of Hematology/Oncology, Department of Medicine, University of California , Los Angeles M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Francis Wright 10 Division of General Internal Medicine, School of Medicine, University of Colorado Anschutz Medical Campus , Aurora, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kimberly de Dios 11 Diabetes Center, University of California San Francisco , San Francisco, USA 12 Division of Endocrinology and Metabolism, Department of Medicine, University of California San Francisco , San Francisco, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Daniel Kim 13 Department of Medicine, Cedars-Sinai Medical Center , Los Angeles, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Eric Mukherjee 14 Department of Dermatology, Vanderbilt University Medical Center , Nashville, USA M.D., Ph.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Eric Mukherjee Matthew S. Krantz 1 Department of Biomedical Informatics, Vanderbilt University Medical Center , Nashville, USA 15 Division of Allergy, Pulmonary, and Critical Care Medicine, Department of Medicine, Vanderbilt University Medical Center , Nashville, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lydia Yao 6 Department of Biostatistics, Vanderbilt University Medical Center , Nashville, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Douglas B. Johnson 16 Department of Medicine, Vanderbilt University Medical Center , Nashville, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Elizabeth J. Phillips 16 Department of Medicine, Vanderbilt University Medical Center , Nashville, USA 17 Institute for Immunology and Infectious Diseases, Murdoch University , Murdoch, Australia M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Justin M. Balko 16 Department of Medicine, Vanderbilt University Medical Center , Nashville, USA Pharm.D., Ph.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Rajat Mohindra 4 F. Hoffmann-La Roche , Basel, Switzerland M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zoe Quandt 11 Diabetes Center, University of California San Francisco , San Francisco, USA 12 Division of Endocrinology and Metabolism, Department of Medicine, University of California San Francisco , San Francisco, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Background Large language models (LLMs) have emerged as transformative technologies, revolutionizing natural language understanding and generation across various domains, including medicine. In this study, we investigated the capabilities, limitations, and generalizability of Generative Pre-trained Transformer (GPT) models in analyzing unstructured patient notes from large healthcare datasets to identify immune-related adverse events (irAEs) associated with the use of immune checkpoint inhibitor (ICI) therapy. Methods We evaluated the performance of GPT-3.5, GPT-4, and GPT-4o models on manually annotated datasets of patients receiving ICI therapy, sampled from two electronic health record (EHR) systems and seven clinical trials. A zero-shot prompt was designed to exhaustively identify irAEs at the patient level (main analysis) and the note level (secondary analysis). The LLM-based system followed a multi-label classification approach to identify any combination of irAEs associated with individual patients or clinical notes. System evaluation was conducted for each available irAE as well as for broader categories of irAEs classified at the organ level. Results Our analysis included 442 patients across three institutions. The most common irAEs manually identified in the patient datasets included pneumonitis (N=64), colitis (N=56), rash (N=32), and hepatitis (N=28). Overall, GPT models achieved high sensitivity and specificity but only moderate positive predictive values, reflecting a potential bias towards overpredicting irAE outcomes. GPT-4o achieved the highest F1 and micro-averaged F1 scores for both patient-level and note-level evaluations. Highest performance was observed in the hematological (F1 range=1.0-1.0), gastrointestinal (F1 range=0.81-0.85), and musculoskeletal and rheumatologic (F1 range=0.67-1.0) irAE categories. Error analysis uncovered substantial limitations of GPT models in handling textual causation, where adverse events should not only be accurately identified in clinical text but also causally linked to immune checkpoint inhibitors. Conclusion The GPT models demonstrated generalizable abilities in identifying irAEs across EHRs and clinical trial reports. Using GPT models to automate adverse event detection in large healthcare datasets will reduce the burden on physicians and healthcare professionals by eliminating the need for manual review. This will strengthen safety monitoring and lead to improved patient care. 1. INTRODUCTION While immune checkpoint inhibitors (ICIs) have greatly enhanced clinical outcomes across multiple cancer types, 1 – 6 their use is often accompanied by immune-related adverse events (irAEs), which occur in 74-90% of patients depending on type of ICI therapy. 7 These events, which can be severe with combination immunotherapies, affect a broad spectrum of organs, including the colon, liver, lungs, heart, nervous system, skin, and endocrine system. 8 – 15 Optimizing the management of irAEs in clinical practice, thereby increasing the safety and effectiveness of cancer immunotherapies, is a major focus in precision oncology research. However, current limitations on the timely and accurate identification of irAEs in clinical settings restrict the implementation of such studies at scale. Major challenges stem from the lack of generalizable methods for identifying all types of toxicities and the difficulties of analyzing dynamically evolving real-world healthcare data (especially unstructured data) such as electronic health records (EHRs) and clinical trials datasets. Previous work by our groups and others has shown that irAEs are poorly captured by structured EHR data like International Classification of Diseases (ICD) codes. 16 – 19 Moreover, specific ICD-10 codes for ICI associated irAEs have only become available recently 20 and it would take time before they are routinely used to abstract irAE data from large healthcare datasets. In contrast, causal relationships between ICI exposures and their irAEs are better encoded in clinical notes, as they allow healthcare providers to document treatment plans, adverse drug reactions, and justifications for discontinuing medications in plain language and with increased detail. 18 , 19 , 21 Still, the automatic identification of irAEs in clinical text poses significant technical challenges since the mention of an irAE in a patient note does not necessarily imply that the patient is currently experiencing the adverse event. For example, an irAE in clinical text may be marked as unlikely or possible, or presented hypothetically as a potential risk during ICI therapy. Further, even when such an event is positively asserted in clinical notes, its underlying cause may not be linked to an ICI exposure. Recent progress of generative artificial intelligence (AI) technologies, such as Generative Pre-trained Transformer (GPT), 22 – 24 underscores the potential of large language models (LLMs) as a transformative approach for identifying irAEs in patient notes. These models have already shown exceptional proficiency in extracting linguistic patterns that convey complex clinical information including clinical phenotypes, assertion statuses, and relationships between medical concepts (e.g., drug–adverse reaction). 25 – 29 In this study, we describe the development and evaluation of irAE-GPT, a holistic system created to automatically identify irAEs in clinical text leveraging the state-of-the-art OpenAI’s GPT models, GPT-3.5, GPT-4, and GPT-4o, and two distinct data sources from multiple institutions: clinical trials and real-world EHR data. Clinical trial datasets summarize safety information–such as irAE symptoms, specific irAE diagnoses, and attributes like therapeutic exposure, treatment duration, and comorbidities– in semi-structured reports using standard terminology. In contrast, real-world EHR data capture a comprehensive, longitudinal record of patients’ health trajectories, encompassing diagnoses, procedures, treatments, medications, encounters, laboratory measurements, clinical notes, images, outcomes and other relevant clinical information over time. The objectives of this study were to: (1) assess the capabilities of GPT models for irAE identification in EHR and clinical trial datasets, (2) explore the limitations of GPT models for this task, and (3) evaluate the generalizability and reproducibility of these models across different data sources originating from multiple institutions. 2. METHODS 2.1 Study design This retrospective cohort study utilizes data collected from two large EHR systems, Vanderbilt University Medical Center (VUMC) and University of California San Francisco (UCSF), and from Roche-sponsored clinical trials. The study population included patients who received any of the following ICI therapies, either as monotherapy or in combination regimens: atezolizumab, avelumab, durvalumab, ipilimumab, nivolumab, or pembrolizumab. The main analysis was conducted at the patient level, where, for each patient in the study, irAE-GPT reviewed their corresponding notes to identify all possible irAEs resulting from the patient’s exposure to ICI therapy. This analysis also involved the aggregation of irAE predictions by GPT models across all the notes of the same patient. A secondary analysis was carried out on VUMC data to assess irAE-GPT’s performance in extracting irAEs caused by ICIs at the note level. As patients may experience none, one, or several irAEs following ICI exposure, a multi-label classification approach was adopted to ensure LLMs can identify any combination of irAEs associated with each patient or clinical note. System evaluation was conducted for each available irAE as well as for broader categories of irAEs classified at organ level. The irAEs corresponding to each irAE category are listed in Table S1 . The institutional review boards at VUMC and UCSF approved this study, and all internal processes were followed to enable secondary use of data from Roche sponsored studies for this analysis. 2.2 Data sources At VUMC, the Synthetic Derivative, a de-identified EHR database of 4 million records, was used to sample patients who underwent ICI therapy between 2009 and 2019. Starting from the first day of ICI administration, the health record data associated with each patient was manually reviewed by clinical experts to label all irAEs experienced by the patient or ‘None’ in the event the patient had no adverse reactions caused by ICI therapy. For each patient, clinical notes timestamped between the first date of ICI exposure and up to six months after the last date of ICI administration were selected for LLM analysis. In addition to patient-level annotation of irAEs, the same labels (i.e., a list of irAEs or ‘None’) were used for note-level annotation on individual notes sampled from 20 patients of the above-mentioned dataset. UCSF Clinical Data Warehouse (CDW), which contains de-identified EHR data captured after 2012 from approximately 6 million unique patients, was used to construct the UCSF cohort. The study team leveraged structured EHR data to identify 617 patients who were admitted to the hospital within 6 months of ICI exposure. Manual chart review narrowed this cohort to 114 subjects who were deemed definitely or probably admitted for hospitalization due to at least one irAE. 30 The clinical notes selected for this study included the admission, first and last history and physical (H&P) exam notes for that admission, first and last consult notes from various specialty departments for that admission, and discharge summaries for each hospitalization. Ultimately, the final UCSF cohort included a total of 70 patients with 74 hospital encounters, each having at least one of the specified types of clinical notes. Each patient was annotated with specific types of irAEs including the irAEs requiring admission and any other concurrent irAEs. Basic demographics and other baseline characteristics of each patient were extracted from structured EHR data (e.g., age, sex, race, etc.) and manual annotations (e.g., cancer types, cancer stage, etc.) Clinical trials data came from seven completed Roche-sponsored clinical studies NCT02450331 (n=809), NCT02031458 (n=659), NCT02108652 (n=429), NCT02486718 (n=507), NCT02008227 (n=613), NCT02302807 (n=467) and NCT03024996 (n=778) in genitourinary, renal and lung cancer indications ( Figure S1 ). irAEs were extracted from the structured database using a broad search of MedDRA preferred terms and this was compared to results from the LLM analysis of free text narratives (written and reviewed by qualified physicians or scientists based on data captured in the study eCRF and additional information provided by the PIs/study sites) to extract irAEs. 2.3 irAE-GPT architecture irAE-GPT is a high-throughput phenotyping system designed to systematically identify irAEs from EHR and clinical trial datasets. It leverages the Azure OpenAI Service, offering access to private, institutionally managed GPT model instances that ensure security and compliance with legal and business agreements. Central to the LLM-based system is a zero-shot prompt crafted to identify irAEs in a clinical note ( Figure 1 ). Specifically, for a given patient note, along with prespecified lists of irAEs and ICI treatments, the prompt systematically queries the LLM to determine whether the note describes the patient experiencing any irAEs caused by exposure to any of their ICI treatments. For easier processing, the LLM output was required to be in JavaScript Object Notation (JSON) format, with each irAE corresponding to a binary response (‘Yes’ or ‘No’) based on its presence in the clinical note. Further, sets of text expressions describing synonyms, acronyms, or semantically related medical concepts–which we called “synsets”–were constructed for selected irAEs to better guide the LLM in the irAE identification process. For example, using the synset information corresponding to ‘ neuropathy’ listed in Table S2 , its augmented query becomes “ Output ‘Yes’ if the patient has experienced neuropathy (neurotox, neurotoxicity) because of exposure to one or more immune checkpoint inhibitors. Otherwise, output ‘No’. ” Notably, the irAE list was adapted for each institution such that each element from the list was annotated at least once in the corresponding dataset. Similarly, multiple site-specific irAE synsets emerged due to variations in the manual annotation processes at each institution. For example, the prompt for the VUMC run had separate queries for colitis and diarrhea whereas the UCSF prompt had only one query for both concepts because they were annotated under the same irAE label. In this case, the colitis synset for the UCSF prompt contained both ‘ colitis ’ and ‘ diarrhea ’. The python code with the implementation of irAE-GPT is available at https://github.com/bejanlab/irAE-GPT.git . Download figure Open in new tab Figure 1 Prompt template for identifying irAEs due to ICI exposure in patient notes. The irAEs used for each dataset are listed in Tables S3-S5 . 2.4 irAE-GPT evaluation The system was deployed at each institution, and its evaluation utilized the complete dataset from the corresponding institution. The same zero-shot prompt and hyperparameters were used for each run except for site-specific irAE lists and synsets. Each LLM was executed with the temperature–a hyperparameter that controls the level of randomness or determinism in LLM output generation–set to 0. This setting is optimal for binary classification tasks, ensuring the model consistently selects the highest-probability token, corresponding to ‘ Yes ’ or ‘ No ’ for each irAE. Performance metrics including precision (positive predictive value or PPV), recall (sensitivity), specificity, and F1 score were computed by comparing the manual annotations against the results extracted by irAE-GPT for each irAE. Patient-level evaluation involved aggregating the LLM results from all notes per patient and comparing the aggregated results with the annotated irAE labels of the corresponding patient. To extract performance measures for irAE categories, we converted both irAE annotations and irAE results generated by the system using the mappings listed in Table S1 . Micro- and macro-averaged results were also computed to assess the overall performance across all irAEs and irAE categories. Because our task emphasized the accurate identification of positive outcomes, F1 and micro-averaged F1 were selected as the primary metrics throughout the experiments. Finally, error analysis was conducted based on manual assessment of clinical notes corresponding to false positive and false negative results. 3. RESULTS 3.1 Description of datasets The cohorts at VUMC, UCSF, and Roche included 100, 70, and 272 patients, respectively ( Table 1 ). The mean age of the patients at ICI initiation across the three cohorts ranged from 62.8 to 65.2 years, with most being male and White. The VUMC and UCSF patients were more likely to be treated with a PD-(L)1 inhibitor followed by a combination of PD-(L)1 and CTLA-4 inhibitors while all Roche patients were treated with a PD-L1 inhibitor. Melanoma was the most common cancer type among VUMC and UCSF patients, while lung cancer was the most frequent in the selected Roche-sponsored clinical trials. Each institute adopted its own strategy in annotating irAEs and selecting unstructured patient notes for LLM-based analysis. At VUMC, the 100-patient cohort was selected from a total of 747 patients who were exposed to ICI therapy and annotated with irAEs ( Table S3 ). Since there were no restrictions imposed on the ICI exposed patients for irAE annotation, many patients were found as not experiencing any irAE (42 patients labeled as ‘None’ out of 100 patients selected for the study). From the patients with at least one irAE, most of them had rash (N=10), colitis (N=9), and pneumonitis (N=8). As no restriction was imposed on patient note selection either (e.g., by note type), 26,432 of them (264 notes per patient, on average) were included for LLM-based irAE identification. At UCSF, all 70 patients were found as experiencing at least one irAE as they were selected for annotation from patients admitted for hospitalization due to definite or probable irAE (hence, with more severe events). Gastrointestinal irAEs, mostly colitis (N=23) and hepatitis (N=15), and pneumonitis (N=15) were the most common irAEs at hospitalization ( Tables 1 and S4 ). For LLM-based irAE identification, the study team curated a set of 487 clinical notes from 74 hospital encounters (∼7 notes per patient) corresponding to the 70 patients. Finally, at Roche, there were 3,471 patients who received ICI therapy in 7 completed trials. From the dataset, 272 patients, who corresponded to the same number of reports, were extracted using a broad structured MedDRA search for possible irAEs. A subset of the most common irAEs were selected for the LLM-based analysis ( Figure S1 , Table S5 ). Notably, some adverse event terms extracted during this process may not qualify as irAEs (e.g., influenza) but might instead be indirectly associated with ICI therapy. For instance, influenza-like symptoms may be reported as influenza. In the Roche study cohort, the most common irAEs were pneumonitis (N=41), colitis (N=24), and rash (N=15). View this table: View inline View popup Download powerpoint Table 1 Selected characteristics of study participants across all sites. 3.2 Patient-level evaluation We adopted simple strategies to aggregate the predictions for all the notes of the same patient. For VUMC results, we balanced the tradeoff between precision and recall by determining the optimal decision threshold—the number of notes with irAE-positive predictions—used to classify patient-level irAE outcomes. The optimal decision threshold values determined by GPT-3.5, GPT-4, and GPT-4o models for maximizing the micro-averaged F1 score were 6, 10, and 18, respectively ( Figures S2-S4 ). The patient-level irAE predictions on UCSF and Roche datasets were determined based on at least one positive prediction (decision threshold=1) in any of the notes corresponding to the same patient. Overall, the GPT models exhibited high recall (sensitivity) and specificity but achieved only moderate precision (positive predictive value, PPV) across datasets from all institutions ( Tables S6-S8 ). For multiple irAEs at VUMC (e.g., hemolytic anemia, lupus flare, bullous pemphigoid), UCSF (e.g., Stevens-Johnson syndrome, Guillain-Barre), and Roche (e.g., bladder tamponade, ascending flaccid paralysis), the GPT models achieved 100% F1 scores. However, the models’ performance was relatively modest for several more common adverse events including arthritis ( F 1 VUMC : GPT -4 = 0.22, F 1 UCSF : GPT -3.5 = 0.27), rash ( F 1 VUMC : GPT -3.5 = 0.31), colitis ( F 1 VUMC : GPT -4 o = 0.32), fever ( F 1 UCSF : GPT -4 o = 0.32), hypothyroidism ( F 1 Roche : GPT -4 = 0.31), and pyrexia ( F 1 Roche : GPT -4 o = 0.33). The overall trends in primary measures, F1 and micro-averaged F1, indicate GPT-4o as the best performing model. Organ-level classification of irAEs facilitates the comparison of results both across institutions and within individual irAE categories ( Figure 2 and Table 2 ). Across the VUMC, UCSF, and Roche datasets, the best results were observed in the hematological (F1 range=1.0-1.0), gastrointestinal (F1 range=0.81-0.85), and musculoskeletal/rheumatologic (F1 range=0.67-1.0) categories, respectively. GPT models showed consistently high F1 scores in the pulmonary category across all datasets. Notably, aggregating manually annotated irAEs and irAE predictions into organ-level irAE categories improves the micro-averaged F1 scores across all datasets ( Tables S6-S8 ). Download figure Open in new tab Figure 2 Comparative analysis of patient-level results across datasets from 3 institutions: Vanderbilt University Medical Center (VUMC), University of California San Francisco (UCSF), and Roche. The radar charts at the top highlight variations in F1 scores across irAE categories, whereas the lollipop plots at the bottom summarize the macro- and micro-averaged F1 scores. Detailed results of this analysis are provided in Table 2 . View this table: View inline View popup Download powerpoint Table 2 Patient-level evaluation for the identification of irAE categories across the datasets from Vanderbilt University Medical Center (VUMC), University of California San Francisco (UCSF), and Roche. All the result values in the table correspond to F1 scores. 3.3 Note-level evaluation A note-level evaluation was conducted to more effectively assess the robustness of GPT models, given that patient-level outcomes depend on the aggregation of irAE predictions across all notes for the same patient. Further, this type of approach facilitates the identification of irAEs over time, offering direct utility for incident-based pharmacoepidemiologic studies. Overall, despite finding optimal decision thresholds for patient-level predictions, higher micro-averaged F1 scores were observed in note-level evaluation compared to patient-level evaluation for identifying both irAEs (micro-averaged F1 ranges, [0.50-0.57] vs. [0.46-0.48]) and irAE categories (micro-averaged F1 ranges, [0.61-0.66] vs. [0.56-0.59]) in the VUMC dataset ( Tables 3 and S6 ). Consistent with the trends shown for patient-level evaluation, GPT-4o yielded the best performing F1 scores for note-level evaluation. View this table: View inline View popup Download powerpoint Table 3 Note-level evaluation for the identification of irAEs and their corresponding categories in the Vanderbilt University Medical Center (VUMC) dataset. 3.4 Error analysis The manual review of notes associated with false positive results revealed that LLMs struggle to solve causal relationships involving positively asserted events with unspecified or unrelated etiological factors. False positive findings also included adverse events characterized by an assertion status indicating hypothetical, uncertain or possible scenarios, or events not associated with the patient. Table S9 lists examples with clinical note excerpts where such adverse events were misclassified by LLMs as irAEs. Finally, several false positives were ultimately confirmed as true irAEs, having been unnoticed during manual annotation (e.g., “ treated with ipilimumab c/b hypophysitis and colitis ”). A significant lower number of false negative results were generated by the GPT models, many of which being improperly described in the zero-shot prompt. For instance, the LLMs failed to associate terms such as “ liver inflammation ” and “ developed grade 2 elevated LFTs ” with hepatitis, as the medical concepts encoded in these expressions were missing from the hepatitis synset. Improper tokenization of specific medical terms may also impact how accurately GPT models recognize irAEs in clinical notes. Subword tokenization examples generated by Tiktoken, 31 the tokenizer developed by OpenAI for its GPT models, include [ ‘neuropathy’ ] → [ ’neu’, ‘rop’, ‘athy’ ], [ ‘neurotox’ ] → [ ’ne’, ‘uro’, ‘to’, ‘x’ ], [‘ myasthenia gravis ’] → [ ’my’, ‘ast’, ‘hen’, ‘ia’, ‘ gr’, ‘avis’ ], and [‘ hypothyroid’ ] → [ ’h’, ‘yp’, ‘othy’, ‘roid’ ]. 4. DISCUSSION This study demonstrates that large language models have potential to effectively identify immune-related adverse events (irAEs) in two distinct textual modalities: clinical notes from two large, tertiary, real-world EHR systems and patient reports from seven clinical trials. Moreover, our findings suggest that GPT models are well-suited for exhaustive identification of irAEs across diverse clinical contexts. At VUMC, we evaluated the LLMs in both inpatient and outpatient settings for patients undergoing ICI therapy, regardless of whether they experienced irAEs. The UCSF experiments utilized EHR data from patients who were hospitalized within six months after ICI exposure due to one or more irAEs, which are considered more severe. The experimental setup at Roche relied on data from patients participating in clinical trials on ICI therapy and experiencing at least one irAE. The primary outcome of our study is to demonstrate irAE-GPT as a generalizable, high-throughput system for automatically extracting therapy-associated toxicities from clinical text. The system aims to alleviate the burden of manually annotating irAEs in clinical text, a labor-intensive task requiring experienced and specialized clinicians. It extracts irAE outcomes at both patient and note levels, enabling a wide range of pharmacoepidemiologic studies, from those focusing on prevalence to those addressing incidence, ultimately contributing to guide clinical care to improve safe and optimized use of novel immunotherapies. The evaluation of the GPT models revealed several significant findings. First, the models achieved consistently high recall/sensitivity and specificity values at the cost of moderate precision scores. This resulted in significantly more false positives than false negatives, indicating a bias in the GPT models toward sensitivity or overpredicting positive outcomes. One of our experiments to mitigate this bias focused on adjusting the decision threshold to optimize the F1 score, achieving a more balanced trade-off between precision and recall. Second, differences in performance were noted among the various types of evaluations conducted. The note-level evaluation yielded better results than the patient-level evaluation, possibly influenced by how note-level predictions were combined to derive patient-level outcomes. Moreover, the prediction of irAE categories demonstrated higher performance values compared to specific irAE evaluation results, primarily due to the relaxed approach in positively predicting the organ-level adverse events. For instance, positively predicting any one of pneumonitis, wheezing, influenza, ARDS, or pleuritis is sufficient to classify the pulmonary category as positive. Finally, the error analysis highlights the challenges faced by LLMs in handling textual causation, a task more complex than named entity recognition, where medical conditions identified in clinical text (e.g., adverse events) must also be linked to specific etiological factors (e.g., immune checkpoint inhibitors). To the best of our knowledge, this is the first study leveraging OpenAI’s GPT models to identify irAEs in clinical text. The closest study to ours is by Sun et al, 19 where they employed the open-source LLM Mistral OpenOrca 32 to detect four irAEs (colitis, hepatitis, myocarditis, and pneumonitis) in electronic health records of hospitalized patients at Massachusetts General Hospital (MGH) and the Brigham and Women’s Hospital (BWH). A straightforward analysis using averaged F1 scores over the four irAEs revealed that Mistral OpenOrca underperformed on the MGH (mean F1=0.26) and BWH (mean F1=0.29) datasets compared to GPT models, which performed better on the VUMC (mean F1=0.60), UCSF (mean F1=0.73), and Roche (mean F1=0.65) datasets. Here, the Roche results reflect the averaged F1 scores for identifying colitis, hepatitis, and pneumonitis, as myocarditis events were absent in the corresponding dataset. With the anticipated release of new ICD-10 codes for ICI-associated irAEs, 20 the classification of irAEs is expected to become more precise, leading to improvements in data availability, comparability, and quality. However, the consistent and routine adoption of these new codes by clinicians, as well as their validation in clinical applications, will require time. The synergy between irAE-GPT and the new ICD codes could enable a systematic approach to understanding the spectrum and burden of ICI-associated irAEs, with far-reaching implications for clinical practice, public health, and research. Our study has several limitations that warrant further investigation in future research work. First, we did not investigate widely-used methods for enhancing the performance of LLMs, including prompt engineering, few-shot learning, and fine-tuning. 23 , 33 , 34 For example, our groups and others have previously demonstrated that LLMs are highly sensitive to the way prompts are constructed, and that prompt augmentation and optimization play a key role in improving the performance of various tasks. 26 , 33 , 35 , 36 Second, due to the input token limit of GPT models, for the patient level approach, we implemented a strategy to aggregate irAE predictions from all notes belonging to the same patient. This is because, although some GPT models offer relatively large token limits, using prompts that could potentially contain all the notes of a patient is still infeasible. For example, on the VUMC dataset, a patient has on average 264 notes during the ICI exposure period. In future work, we intend to use retrieval-augmented generation (RAG) techniques 37 as an alternative approach, creating one prompt per patient that contains only irAE-relevant text snippets extracted from the entire patient record. Third, the possibility that LLMs may perpetuate and exacerbate health disparities in clinical studies of cancer immunotherapy safety and effectiveness has not yet been explored. While not within the scope of our study, the implementation of fairness assessments and bias mitigation techniques is critical for ensuring transparent and equitable LLM approaches in such studies. Lastly, the lack of a standardized and comprehensive list of irAEs, combined with challenges in categorizing them through splitting or grouping, may limit the applicability of our findings to other institutions or treatment centers. 5. CONCLUSION This study introduces irAE-GPT, a high-throughput phenotyping system that leverages GPT models to systematically identify irAEs in unstructured patient notes from multiple EHR systems and clinical trials. The GPT models demonstrated robust performance and generalizable capabilities in identifying irAEs across diverse clinical contexts, reducing the need for manual annotation of adverse events in clinical text and highlighting their potential to improve the characterization of the safety profile of cancer immunotherapies. Efforts should continue to enhance the capabilities of these models in identifying textual causation between medication exposures and adverse events, while also addressing their potential biases in extraction of safety data from clinical text. irAE-GPT is publicly available, allowing researchers to experiment with their local healthcare datasets. Data availability The source code used in the analysis is publicly available on GitHub at https://github.com/bejanlab/irAE-GPT.git . The summary statistics extracted for this study are provided in the manuscript and supplementary material. Any request to access the Synthetic Derivative data will need to be reviewed and approved by Vanderbilt University Medical Center. Researchers will need to provide evidence of IRB approval for their study. For the approved studies, data will be released via a Data Use Agreement. Funding CAB, DBJ, and JB are supported by R01CA227481. CAB and JB are supported by R01HL156021. CAB is supported in part by R21HD113234. The UCSF team is supported by the Research Allocation Program grant. JS is supported by the NIH T32 fellowship (PCORT UroGynCan T32CA251072). EJP receives funding from R01HG010863, R01AI152183, U01AI154659, the NHMRC of Australia, SJS Research Fund and the Angela Anderson Research fund. ZQ is supported by the NIH NIDDK DiabDocs K12DK133995 and a Larry L Hillblom Foundation Start Up Grant. The Vanderbilt University Medical Center dataset used in this study is supported by numerous funding sources including UL1TR002243, UL1TR000445, and UL1RR024975. The analysis of Roche data included completed clinical trials funded by Roche. Disclosures DBJ has served on advisory boards or as a consultant for AstraZeneca, BMS, The Jackson Laboratory, Merck, Mosaic ImmunoEngineering, Novartis, Pfizer, and Teiko, and has received research funding from BMS and Incyte, and has patents pending for use of MHC-II as a biomarker for immune checkpoint inhibitor response, and abatacept as treatment for immune-related adverse events. EJP has served as a consultant for Janssen, Rapt, Servier, Espirion, Verve, Elion and UpToDate outside of the submitted work, and receives royalties from UpToDate. GSC and RM are employees and shareholders of Roche. They received support for preparation of this manuscript, and are co-inventors on patents filed by Roche related to use of atezolizumab. ZQ has consulted for Novartis and Sanofi. Acknowledgements We thank the staff of the Vanderbilt’s Institute for Clinical and Translational Research (VICTR) and HealthIT at Vanderbilt University Medical Center (VUMC) for facilitating access to the Synthetic Derivative and VUMC-managed generative AI tools. We thank the staff of the Bakar Computational Health Sciences Institute, UCSF Information Commons, and the Center for Data-driven Insights and Innovation within the University of California Health system, as well as the UCSF AI Tiger Team, Academic Research Services, Research Information Technology, and the Chancellor’s Task Force for Generative AI for their software development, analytical and technical support related to the use of Versa API gateway (the UCSF secure implementation of large language models and generative AI via API gateway), Versa chat (the chat user interface), and related data asset and services. We would also like to acknowledge patients, families, investigators, and site staff who performed the Roche-sponsored clinical trials that were included in this analysis. Significant contributions were made by Christian W. Thorball and He Xu, both from Precision Medicine Unit, Lausanne University Hospital and University of Lausanne, Lausanne, Switzerland. Footnotes ↵ ++ Supervised this work jointly REFERENCES 1. ↵ Hodi FS , O’Day SJ , McDermott DF , et al. Improved survival with ipilimumab in patients with metastatic melanoma . N Engl J Med . Aug 19 2010 ; 363 ( 8 ): 711 – 23 . doi: 10.1056/NEJMoa1003466 OpenUrl CrossRef PubMed 2. Robert C , Long GV , Brady B , et al. Nivolumab in Previously Untreated Melanoma without BRAF Mutation . N Engl J Med . Jan 22 2015 ; 372 ( 4 ): 320 – 330 . doi: 10.1056/NEJMoa1412082 OpenUrl CrossRef PubMed 3. Larkin J , Chiarion-Sileni V , Gonzalez R , et al. Combined Nivolumab and Ipilimumab or Monotherapy in Untreated Melanoma . N Engl J Med . Jul 2 2015 ; 373 ( 1 ): 23 – 34 . doi: 10.1056/NEJMoa1504030 OpenUrl CrossRef PubMed 4. Larkin J , Chiarion-Sileni V , Gonzalez R , et al. Five-Year Survival with Combined Nivolumab and Ipilimumab in Advanced Melanoma . N Engl J Med . Oct 17 2019 ; 381 ( 16 ): 1535 – 1546 . doi: 10.1056/NEJMoa1910836 OpenUrl CrossRef PubMed 5. Tawbi HA , Schadendorf D , Lipson EJ , et al. Relatlimab and Nivolumab versus Nivolumab in Untreated Advanced Melanoma . N Engl J Med . Jan 6 2022 ; 386 ( 1 ): 24 – 34 . doi: 10.1056/NEJMoa2109970 OpenUrl CrossRef PubMed 6. ↵ Postow MA , Chesney J , Pavlick AC , et al. Nivolumab and ipilimumab versus ipilimumab in untreated melanoma . N Engl J Med . May 21 2015 ; 372 ( 21 ): 2006 – 17 . doi: 10.1056/NEJMoa1414428 OpenUrl CrossRef PubMed 7. ↵ Brahmer JR , Abu-Sbeih H , Ascierto PA , et al. Society for Immunotherapy of Cancer (SITC) clinical practice guideline on immune checkpoint inhibitor-related adverse events . J Immunother Cancer . Jun 2021 ; 9 ( 6 ) doi: 10.1136/jitc-2021-002435 OpenUrl Abstract / FREE Full Text 8. ↵ Johnson DB , Balko JM , Compton ML , et al. Fulminant Myocarditis with Combination Immune Checkpoint Blockade . N Engl J Med . Nov 3 2016 ; 375 ( 18 ): 1749 – 1755 . doi: 10.1056/NEJMoa1609214 OpenUrl CrossRef PubMed 9. Moslehi JJ , Salem JE , Sosman JA , Lebrun-Vignes B , Johnson DB . Increased reporting of fatal immune checkpoint inhibitor-associated myocarditis . Lancet . Mar 10 2018 ; 391 ( 10124 ): 933 . doi: 10.1016/S0140-6736(18)30533-6 OpenUrl CrossRef PubMed 10. Postow MA , Sidlow R , Hellmann MD . Immune-Related Adverse Events Associated with Immune Checkpoint Blockade . N Engl J Med . Jan 11 2018 ; 378 ( 2 ): 158 – 168 . doi: 10.1056/NEJMra1703481 OpenUrl CrossRef PubMed 11. Wang DY , Salem JE , Cohen JV , et al. Fatal Toxic Effects Associated With Immune Checkpoint Inhibitors: A Systematic Review and Meta-analysis . Jama Oncol . Dec 1 2018 ; 4 ( 12 ): 1721 – 1728 . doi: 10.1001/jamaoncol.2018.3923 OpenUrl CrossRef PubMed 12. Fahey CC , Gracie TJ , Johnson DB . Immune checkpoint inhibitors: maximizing benefit whilst minimizing toxicity . Expert Rev Anticancer Ther . May 22 2023 : 1 – 11 . doi: 10.1080/14737140.2023.2215435 OpenUrl CrossRef 13. Johnson DB , Nebhan CA , Moslehi JJ , Balko JM . Immune-checkpoint inhibitors: long-term implications of toxicity . Nat Rev Clin Oncol . Apr 2022 ; 19 ( 4 ): 254 – 267 . doi: 10.1038/s41571-022-00600-w OpenUrl CrossRef PubMed 14. Das S , Johnson DB . Immune-related adverse events and anti-tumor efficacy of immune checkpoint inhibitors . J Immunother Cancer . Nov 15 2019 ; 7 ( 1 ): 306 . doi: 10.1186/s40425-019-0805-8 OpenUrl Abstract / FREE Full Text 15. ↵ Salem JE , Manouchehri A , Moey M , et al. Cardiovascular toxicities associated with immune checkpoint inhibitors: an observational, retrospective, pharmacovigilance study . Lancet Oncol . Dec 2018 ; 19 ( 12 ): 1579 – 1589 . doi: 10.1016/S1470-2045(18)30608-9 OpenUrl CrossRef PubMed 16. ↵ Nashed A , Zhang S , Chiang CW , et al. Comparative assessment of manual chart review and ICD claims data in evaluating immunotherapy-related adverse events . Cancer Immunol Immunother . Oct 2021 ; 70 ( 10 ): 2761 – 2769 . doi: 10.1007/s00262-021-02880-0 OpenUrl CrossRef PubMed 17. Cheung YM , Hamnvik OR , Shariff A , Gallagher EJ . Dearth of ICD Codes for Complications of Immune Checkpoint Inhibitors Impedes Clinical Care and Research . J Endocr Soc . Feb 9 2023 ; 7 ( 4 ): bvad019 . doi: 10.1210/jendso/bvad019 OpenUrl CrossRef 18. ↵ Barman H , Venkateswaran S , Santo AD , et al. Identification and Characterization of Immune Checkpoint Inhibitor-Induced Toxicities From Electronic Health Records Using Natural Language Processing . JCO Clin Cancer Inform . Apr 2024 ; 8 : e2300151 . doi: 10.1200/CCI.23.00151 OpenUrl CrossRef 19. ↵ Sun VH , Heemelaar JC , Hadzic I , et al. Enhancing Precision in Detecting Severe Immune-Related Adverse Events: Comparative Analysis of Large Language Models and International Classification of Disease Codes in Patient Records . J Clin Oncol . Sep 3 2024 : JCO2400326 . doi: 10.1200/JCO.24.00326 OpenUrl CrossRef 20. ↵ Mittal K , Reynolds KL , Funchain P. Introducing ASPIRE and STORIES: A New International Initiative for Faculty Collaboration and Patient Advocacy in Immune-Related Adverse Events . The ASCO Post . https://ascopost.com/issues/june-10-2024/a-new-international-initiative-for-faculty-collaboration-and-patient-advocacy-in-immune-related-adverse-events/ 21. ↵ Gupta S , Belouali A , Shah NJ , Atkins MB , Madhavan S . Automated Identification of Patients With Immune-Related Adverse Events From Clinical Notes Using Word Embedding and Machine Learning . JCO Clin Cancer Inform . May 2021 ; 5 : 541 – 549 . doi: 10.1200/CCI.20.00109 OpenUrl CrossRef PubMed 22. ↵ OpenAI . Improving Language Understanding by Generative Pre-Training . 2018 ; 23. ↵ Brown TB , Mann B , Ryder N , et al. Language models are few-shot learners. presented at: Proceedings of the 34th International Conference on Neural Information Processing Systems ; 2020 ; Vancouver, BC, Canada . 24. ↵ OpenAI. GPT-4 Technical Report . 2023 : arXiv : 2303.08774 . doi: 10.48550/arXiv.2303.08774 Accessed March 01, 2023 . https://ui.adsabs.harvard.edu/abs/2023arXiv230308774O OpenUrl CrossRef 25. ↵ Alsentzer E , Rasmussen MJ , Fontoura R , et al. Zero-shot interpretable phenotyping of postpartum hemorrhage using large language models . NPJ Digit Med . Nov 30 2023 ; 6 ( 1 ): 212 . doi: 10.1038/s41746-023-00957-x OpenUrl CrossRef PubMed 26. ↵ Bejan CA , Reed AM , Mikula M , et al. Large language models improve the identification of emergency department visits for symptomatic kidney stones . Sci Rep . Jan 28 2025 ; 15 ( 1 ): 3503 . doi: 10.1038/s41598-025-86632-5 OpenUrl CrossRef PubMed 27. Ji Y , Yu Z , Wang Y. Assertion Detection Large Language Model In-context Learning LoRA Fine-tuning . 2024 : arXiv : 2401.17602 . doi: 10.48550/arXiv.2401.17602 Accessed January 01, 2024 . https://ui.adsabs.harvard.edu/abs/2024arXiv240117602J OpenUrl CrossRef 28. Gu Y , Zhang S , Usuyama N , et al. Distilling Large Language Models for Biomedical Knowledge Extraction: A Case Study on Adverse Drug Events . 2023 : arXiv : 2307.06439 . doi: 10.48550/arXiv.2307.06439 Accessed July 01, 2023 . https://ui.adsabs.harvard.edu/abs/2023arXiv230706439G OpenUrl CrossRef 29. ↵ Li Y , Li J , He J , Tao C . AE-GPT: Using Large Language Models to extract adverse events from surveillance reports-A use case with influenza vaccine adverse events . Plos One . 2024 ; 19 ( 3 ): e0300919 . doi: 10.1371/journal.pone.0300919 OpenUrl CrossRef PubMed 30. ↵ Silverstein J , Wright F , Wang M , et al. Evaluating Survival After Hospitalization Due to Immune-Related Adverse Events From Checkpoint Inhibitors . The Oncologist . 2023 ; 28 ( 10 ): e950 – e959 . doi: 10.1093/oncolo/oyad135 OpenUrl CrossRef PubMed 31. ↵ OpenAI . Tiktoken . Accessed May 25, 2024 , https://github.com/openai/tiktoken 32. ↵ Lab A. Mistral-7B-OpenOrca . Accessed December 03, 2024 , https://huggingface.co/Open-Orca/Mistral-7B-OpenOrca 33. ↵ Perez E , Kiela D , Cho K . True Few-Shot Learning with Language Models . Advances in Neural Information Processing Systems 34 (Neurips 2021) . 2021 ; 34. ↵ Liu PF , Yuan WZ , Fu JL , Jiang ZB , Hayashi H , Neubig G. Pre-train, Prompt, and Predict: A Systematic Survey of Prompting Methods in Natural Language Processing . Acm Computing Surveys . Sep 2023 ; 55 ( 9 )doi:Artn 195 10.1145/3560815 OpenUrl CrossRef 35. ↵ Jiang LY , Liu XC , Nejatian NP , et al. Health system-scale language models are all-purpose prediction engines . Nature . Jun 7 2023 ; doi: 10.1038/s41586-023-06160-y OpenUrl CrossRef 36. ↵ Webson A , Pavlick E . Do Prompt-Based Models Really Understand the Meaning of Their Prompts? Naacl 2022: The 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies . 2022 : 2300 – 2344 . 37. ↵ Lewis P , Perez E , Piktus A , et al. Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks . Advances in Neural Information Processing Systems 33 (Nips 2020) . 2020 ; View the discussion thread. Back to top Previous Next Posted March 06, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following irAE-GPT: Leveraging large language models to identify immune-related adverse events in electronic health records and clinical trial datasets Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share irAE-GPT: Leveraging large language models to identify immune-related adverse events in electronic health records and clinical trial datasets Cosmin A. Bejan , Michelle Wang , Sriram Venkateswaran , Ewa A. Bergmann , Laura Hiles , Yaomin Xu , G Scott Chandler , Sam Brondfield , Jordyn Silverstein , Francis Wright , Kimberly de Dios , Daniel Kim , Eric Mukherjee , Matthew S. Krantz , Lydia Yao , Douglas B. Johnson , Elizabeth J. Phillips , Justin M. Balko , Rajat Mohindra , Zoe Quandt medRxiv 2025.03.05.25323445; doi: https://doi.org/10.1101/2025.03.05.25323445 Share This Article: Copy Citation Tools irAE-GPT: Leveraging large language models to identify immune-related adverse events in electronic health records and clinical trial datasets Cosmin A. Bejan , Michelle Wang , Sriram Venkateswaran , Ewa A. Bergmann , Laura Hiles , Yaomin Xu , G Scott Chandler , Sam Brondfield , Jordyn Silverstein , Francis Wright , Kimberly de Dios , Daniel Kim , Eric Mukherjee , Matthew S. Krantz , Lydia Yao , Douglas B. Johnson , Elizabeth J. Phillips , Justin M. Balko , Rajat Mohindra , Zoe Quandt medRxiv 2025.03.05.25323445; doi: https://doi.org/10.1101/2025.03.05.25323445 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (569) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4442) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1511) Epidemiology (15230) Forensic Medicine (30) Gastroenterology (1126) Genetic and Genomic Medicine (6610) Geriatric Medicine (668) Health Economics (998) Health Informatics (4542) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15923) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6607) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1146) Occupational and Environmental Health (957) Oncology (3337) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (664) Pediatrics (1693) Pharmacology and Therapeutics (692) Primary Care Research (712) Psychiatry and Clinical Psychology (5448) Public and Global Health (9238) Radiology and Imaging (2202) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (596) Sexual and Reproductive Health (714) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a01d799cfc6b52ad',t:'MTc3OTgwNTc5Nw=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.