Full text
48,693 characters
· extracted from
preprint-html
· click to expand
The Expertise Paradox: Who Benefits from LLM-Assisted Brain MRI Differential Diagnosis? | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search The Expertise Paradox: Who Benefits from LLM-Assisted Brain MRI Differential Diagnosis? Severin Schramm , Bastien Le Guellec , Marlene Topka , Mortimer Svec , Paul Backhaus , Viktor Maria Eisenkolb , Evamaria O. Riedel , Mirjam Beyrle , Paul-Sören Platzek , Constanze Ramschütz , Karolin J. Paprottka , Martin Renz , Jannis Bodden , View ORCID Profile Jan S. Kirschke , Sebastian Ziegelmayer , View ORCID Profile Felix Busch , Marcus R. Makowski , Lisa Adams , Keno Bressem , Dennis M. Hedderich , View ORCID Profile Benedikt Wiestler , View ORCID Profile Su Hwan Kim doi: https://doi.org/10.1101/2025.10.28.25338816 Severin Schramm 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Bastien Le Guellec 2 Department of Neuroradiology, Lille University Hospital , Lille, France Find this author on Google Scholar Find this author on PubMed Search for this author on this site Marlene Topka 3 Department of Neurology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mortimer Svec 3 Department of Neurology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Paul Backhaus 4 Department of Neurosurgery, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Viktor Maria Eisenkolb 4 Department of Neurosurgery, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Evamaria O. Riedel 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mirjam Beyrle 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Paul-Sören Platzek 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Constanze Ramschütz 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Karolin J. Paprottka 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Martin Renz 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jannis Bodden 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jan S. Kirschke 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jan S. Kirschke Sebastian Ziegelmayer 5 Department of Diagnostic and Interventional Radiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Felix Busch 5 Department of Diagnostic and Interventional Radiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Felix Busch Marcus R. Makowski 5 Department of Diagnostic and Interventional Radiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lisa Adams 5 Department of Diagnostic and Interventional Radiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Keno Bressem 5 Department of Diagnostic and Interventional Radiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany 6 Department of Cardiovascular Radiology and Nuclear Medicine, German Heart Center Munich, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Dennis M. Hedderich 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Benedikt Wiestler 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany 7 AI for Image-Guided Diagnosis and Therapy, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Benedikt Wiestler Su Hwan Kim 1 Department of Diagnostic and Interventional Neuroradiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany 5 Department of Diagnostic and Interventional Radiology, TUM University Hospital, School of Medicine and Health, Technical University of Munich , Munich, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Su Hwan Kim For correspondence: suhwan.kim{at}tum.de Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Purpose To evaluate how reader experience influences the diagnostic benefit from LLM assistance in brain MRI differential diagnosis. Materials and Methods Neuroradiologists (n = 4), radiology residents (n = 4), and neurology/neurosurgery residents (n = 4) were recruited. A dataset of complex brain MRI cases was curated from the local imaging database (n = 40). For each case, readers provided a textual description of the main imaging finding and their top three differential diagnoses (“Unassisted”). Three state-of-the-art large language models (GPT-4.1, Gemini 2.5 Pro, DeepSeek-R1) were prompted to generate top-three differentials based on the clinical case description and reader-specific findings. Readers then revised their differential diagnoses after reviewing GPT-4.1 suggestions (“Assisted”). To evaluate the association between reader experience and diagnostic benefit, a cumulative link mixed model (CLMM) was fitted, with change in diagnostic result as ordinal outcome, reader experience as predictor, and random intercepts for rater and case. Results LLM-generated differential diagnoses achieved the highest top-3 accuracy when provided with image descriptions from neuroradiologists (top-3: 78.8-83.8%), followed by radiology residents (top-3: 71.8-77.6%), and neurology/neurosurgery residents (top-3: 62.6-64.5%). In contrast, mean relative gains in top-3 accuracy through LLM assistance diminished with increasing experience, with +19.2% for neurology/neurosurgery residents (from 43.2% to 62.6%), +14.7% for radiology residents (from 59.6% to 74.4%), and +4.4% for neuroradiologists (from 83.1% to 87.5%). The CLMM demonstrated a significant negative association between reader experience and diagnostic benefit from LLM assistance (β = −0.10, p = 0.005). Conclusion With increasing reader experience, absolute diagnostic LLM performance with reader-generated input improved, while relative diagnostic gains through LLM assistance paradoxically diminished. Our findings call attention to the divergence between standalone LLM performance and clinically relevant reader benefit, and emphasize the need to account for human-AI interaction in this context. Introduction In recent years, various applications of large language models (LLMs) in radiology have been demonstrated, including study protocoling [ 1 , 2 ], the generation of an impression section of a radiology report [ 3 , 4 ], and data extraction from free-text reports [ 5 – 7 ]. In addition, numerous studies have investigated the potential of LLMs in radiological differential diagnosis, either as standalone tools generating diagnoses from textual imaging findings and key images [ 8 – 11 ], or as assistive tools offering diagnostic suggestions for consideration by human readers [ 12 – 15 ]. Potentially owing to heterogeneity in reader cohorts and study designs, this latter group of studies produced mixed results, with some studies reporting improved accuracy [ 12 , 13 ] while others did not [ 14 , 15 ]. Importantly, human-AI interaction is a crucial factor influencing the diagnostic benefit readers can derive from LLM assistance. A recent large-scale investigation found that while LLM assistance enhanced the diagnostic accuracy of board-certified internal medicine physicians based on clinical case vignettes, the same LLM alone remained superior, indicating that the human experts frequently rejected even correct LLM suggestions [ 16 ]. Similarly, radiology residents were reported to reject correct LLM-suggested diagnoses in 17.9% of cases [ 12 ], underlining the critical challenge of effectively validating LLM outputs. In addition, the diagnostic performance of LLMs is mediated by the quality of its inputs. Radiologist-generated textual imaging findings were found to be the primary determinant of LLM accuracy [ 17 ], indicating that LLM performance may fluctuate substantially with variation in reader inputs. Against this background, we aimed to investigate how reader experience influences the diagnostic benefit from LLM assistance in brain MRI differential diagnosis. We hypothesized that expert-written imaging findings would better encode discriminative features and yield more accurate LLM-generated differential diagnoses than those of novice readers, but that the diagnostic benefit of LLM assistance would be offset by the higher baseline performance of expert readers. Methods This study was approved by ethics committee of the Technical University of Munich, and the need for informed consent was waived. Dataset 40 brain MRI cases from the local imaging database were included. These cases were selected from a local neuroradiology case collection (n = 778). Inclusion criteria were a confirmed diagnosis, established either histopathologically or by independent consensus of at least two neuroradiologists, based on all available clinical and follow-up information. Additionally, each case had to be of sufficient diagnostic complexity to be considered appropriate for use in neuroradiology subspecialty examinations, as determined by two board-certified neuroradiologists (B.W. and D.M.H., each with 10 years of experience). Exclusion criteria included imaging of a modality or anatomical region other than brain MRI, inadequate image quality, or an unconfirmed diagnosis. Conditions spanned the entire neuroradiological spectrum, including vascular (e.g. Moyamoya syndrome), infectious (e.g. toxoplasmosis), neoplastic (e.g. chondrosarcoma), degenerative (e.g. Huntington’s disease), metabolic (e.g. Wernicke encephalopathy), demyelinating (e.g. progressive multifocal leukoencephalopathy), traumatic (e.g. diffuse axonal injury), and developmental (focal cortical dysplasia) disorders. Scans were obtained between January 1, 2009, and April 30, 2024. All cases have been published previously [ 17 ]. Yet, the clinical case descriptions were published only in January 2025, after the knowledge cutoff date of all models used in this study, thereby excluding the possibility of performance overestimation due to data leakage. To prevent potential bias, readers were instructed a priori to report any cases they recognized from prior clinical involvement, and these were excluded from the analysis to ensure that all remaining cases were unfamiliar to the readers. An overview of clinical cases with respective medical histories is provided in Supplement 1 . View this table: View inline View popup Supplement 1: Case overview. Study Design An overview of the study design is shown in Figure 1 . Neuroradiologists (n = 4), radiology residents (n = 4), as well as neurology and neurosurgery residents (hereinafter referred to as “NL/NS residents”, n = 4; n = 2 each) were recruited from a single academic center. Based on the MRI scan, a condensed medical history and patient demographics, readers created a textual description for the main image finding, which was clearly annotated with one or more arrows. In addition, readers indicated a ranked list of up to three differential diagnoses, along with an overall confidence score on a 5-point Likert scale. Brain MRI images were accessed using the local PACS system. Textual MRI findings were then provided to three LLMs (Gemini 2.5 Pro, GPT-4.1, DeepSeek-R1) which were prompted to equally generate the top-3 differential diagnoses. Subsequently, readers reviewed the differential diagnoses of one representative LLM (GPT-4.1), generated from their own textual findings, and provided their final top-three diagnoses and confidence level taking into consideration the model’s suggestions and explanations. GPT-4.1 was selected at random after a preliminary analysis demonstrated comparable diagnostic accuracy among the three tested models. Cases were presented to the readers in a randomized order, which was identical across the unassisted and LLM-assisted reading conditions for each reader. To avoid additional confounding factors from external research, readers were instructed to refrain from consulting textbooks or online resources and to assess the cases solely based on their existing knowledge. Primary outcome was case-level change in top-3 accuracy from unassisted to assisted reading. Secondary outcomes included correctness and completeness ratings of reader-generated imaging findings. Download figure Open in new tab Figure 1: Study Design. Readers first generated a textual description of the main imaging finding and provided their top three differential diagnoses (unassisted). Gemini 2.5 Pro, GPT-4.1, and DeepSeek-R1 were then prompted to generate their top three differential diagnoses based on a condensed medical history and each reader’s finding description. Subsequently, readers reviewed GPT-4.1’s differential diagnoses derived from their own descriptions and provided their final, integrated top three differential diagnoses (assisted). Icons were obtained from flaticon.com. Sample brain MRI image was obtained from: https://doi.org/10.53347/rID-10047 . N: Neurology. NS: Neurosurgery. RAD: Radiology. DDx: differential diagnosis. LLM Setup Gemini 2.5 Pro (“gemini-2.5-pro-preview-03-25”, proprietary model; Google DeepMind, London, UK), GPT-4.1 (“gpt-4.1-2025-04-14”, proprietary model; OpenAI, Inc., San Francisco, USA), and DeepSeek-R1 (“deepseek-r1”, open-source; Hangzhou DeepSeek Artificial Intelligence Basic Technology Research Co., Ltd., Hangzhou, China) were used as representative state-of-the-art LLMs at the time of the study. Gemini 2.5 Pro and GPT-4.1 were accessed through the official APIs of Google ( https://ai.google.dev/api ) and OpenAI ( https://platform.openai.com/docs/models ), respectively. DeepSeek-R1 was accessed via Fireworks AI ( https://fireworks.ai/models ), a generative AI inference platform with servers located in the United States and Europe. Models were prompted as follows: “You are an experienced neuroradiologist. Below is a case presentation including patient demographics, relevant clinical history, and brain MRI findings. Based on this information, provide your top three differential diagnoses ranked in order of likelihood . CASE: {case_description} For each differential diagnosis, briefly explain your reasoning.” The imaging findings were presented to the LLM in German, the native language of the readers. No local fine-tuning or modification was performed. To ensure deterministic outputs, the temperature hyperparameter was set to 0 for all three models. No random seeds beyond the default API configurations were applied. All three models were configured to generate outputs in structured JSON format to ensure consistent and machine-readable responses. Queries were performed on 3 May 2025. Model outputs were provided to readers without any post-processing. Assessment of Diagnostic Accuracy and Image Descriptions In general, only responses indicating the exact pathologic entity were counted as correct (e.g. “Alzheimer’s disease” was counted as incorrect in a case of posterior cortical atrophy). Edge cases in which the response indicated a correct but less specific diagnosis or related but not identical diagnosis were adjudicated by two board-certified neuroradiologists in consensus (B.W. and D.M.H.) ( Supplement 2 ). View this table: View inline View popup Download powerpoint Supplement 2: Adjudication of edge cases. A board-certified neuroradiologist with six years of experience (B.L.G.), who did not participate as a reader and was blinded to the readers’ experience levels but had access to the reference diagnoses and additional online resources (e.g., www.radiopaedia.org ), rated the imaging descriptions for correctness and completeness using a 4-point Likert scale (see Supplement 3 for rating criteria). View this table: View inline View popup Download powerpoint Supplement 3: Legend for expert ratings of reader-generated imaging findings (4-point Likert scale). Top-3 diagnostic accuracy by reader group (proportion of cases where the correct diagnosis was included in the top 3 diagnoses) was determined for unassisted readers, LLMs with reader input, and GPT-4.1-assisted readers. Additionally, the diagnostic benefit of readers was determined by subtracting top-3 accuracy of unassisted readers from top-3 accuracy of assisted readers. Statistical Analysis Statistical analyses were performed using RStudio (version 2025.05.1+153; Posit Software, Boston, MA) and R (version 4.5.1; R Foundation for Statistical Computing, Vienna, Austria). Mixed-effects modeling was conducted using the lme4 and ordinal packages. Statistically significant difference was set at p < .05. To examine how radiological experience modulates the benefit derived from LLM assistance, we fitted a cumulative link mixed model (CLMM), with change in top-3 accuracy (coded as -1, 0, or 1) as the ordinal outcome variable, and radiological experience in years as predictor. To investigate whether radiologist experience influenced the quality of prompts submitted to LLMs, we constructed two additional CLMMs. One model examined prompt correctness as a function of radiological experience in years, while the other assessed prompt completeness (both on a 4-point Likert scale). For all three models, random intercepts were included for both rater and case to account for individual differences among raters and variability across clinical cases. Code Availability Our Python code for generating differential diagnoses via GPT-4.1, DeepSeek-R1, and Gemini 2.5 Pro is publicly available in our GitHub repository at https://github.com/shk03/expertise_paradox . Results Reader Characteristics The neurology and neurosurgery residents did not have any formal radiological training, but a mean experience of 2.4 ± 0.5 years in their respective specialty. Radiology residents had an average of 1.6 ± 0.4 years of overall radiology experience, all of which was dedicated neuroradiology training. Neuroradiologists had 11.8 ± 4.9 years of overall radiology experience, including 6.5 ± 3.7 years in neuroradiology ( Table 1 ). View this table: View inline View popup Download powerpoint Table 1: Reader characteristics. NL/NS: Neurology/Neurosurgery. RAD: Radiology. Diagnostic Performance 9 out of 480 case readings were excluded due to the familiarity of the readers with the case, as disclosed by the readers. Two sample cases including the case details, reader findings, and differential diagnoses are shown in Figures 2 , 3 . Download figure Open in new tab Figure 2: Sample case. The correct diagnosis was clivus chordoma. While GPT-4.1 suggested the correct diagnosis based on imaging findings from both the neurology resident and the neuroradiologist, only the neurology resident benefited from it, though his diagnostic confidence remained very low. Imaging descriptions were translated from German to English for illustration purposes. Download figure Open in new tab Figure 3: Sample case. The correct diagnosis was subependymal giant cell astrocytoma (SEGA). Although GPT-4.1 suggested the correct diagnosis based on imaging findings provided by the neuroradiologist, neither of the two readers benefited from it. Imaging descriptions were translated from German to English for illustration purposes. For all three models, LLM-generated differential diagnoses achieved the highest top-3 accuracy when provided with image descriptions from neuroradiologists (top-3: 78.8 [126/160] - 83.8% [134/160]), followed by radiology residents (top-3: 71.8 [112/156] - 77.6% [121/156]), and NL/NS residents (top-3: 62.6 [97/155] - 64.5% [100/155]) ( Figure 4 ). Throughout reader groups, Gemini 2.5 Pro demonstrated superior performance (NL/NS residents: 64.5% [100/155], radiology residents: 77.6% [121/156], neuroradiologists: 83.8% [134/160]). Download figure Open in new tab Figure 4: Diagnostic accuracy. The light blue area highlights the gain in diagnostic accuracy of readers through GPT-4.1 assistance. N: Neurology. NS: Neurosurgery. Gains in top-3 accuracy from GPT-4.1 assistance diminished with increasing experience, with +19.2% for neurology/neurosurgery residents (from 43.2% [70/155] to 62.6% [97/155]),+14.7% for radiology residents (from 59.6% [93/156] to 74.4% [116/156]), and +4.4% for neuroradiologists (from 83.1% [133/160] to 87.5% [140/160]). The CLMM revealed a significant negative association between reader experience and diagnostic benefit from LLM assistance (β = -0.098, SE = 0.035, z = -2.788, p = 0.005). This indicates that each additional year of radiological experience was associated with decreased benefit from LLM assistance. The model included random effects for both case (variance = 0.90, SD = 0.95) and rater (variance = 0.21, SD = 0.46), demonstrating substantial variation across cases and moderate variation across individual raters. Figure 5 depicts the transitions in reader responses - from correct to incorrect and vice versa - when assisted by GPT-4.1. Across reader groups, LLM assistance resulted in a small number of correct responses being changed to incorrect, but these were outweighed by a greater number of incorrect responses corrected, yielding an overall diagnostic benefit. Only a small proportion of cases involved readers abandoning a correct diagnosis in favor of an incorrect LLM suggestion (NL/NS residents: 2.6% [4/155], radiology residents: 1.9% [3/156],neuroradiologists: 1.9% [3/160]). More commonly, readers failed to adopt correct LLM suggestions (NL/NS residents: 4.5% [7/155], radiology residents: 3.2% [5/156], neuroradiologists: 3.8% [6/160]). Download figure Open in new tab Figure 5: Sankey diagram illustrating the transition from correct to incorrect responses and vice versa. The first row represents the progression from unassisted readers, through GPT-4.1, to assisted readers. The second row illustrates the direct transition from unassisted to assisted readers. Throughout all reader groups, GPT-4.1 assistance led to a small number of responses changing from correct to incorrect. However, this number was smaller than the number of responses changing from incorrect to correct, resulting in an overall diagnostic benefit. Confidence Levels Mean diagnostic confidence increased with GPT-4.1 assistance across all rater groups, although the magnitude of improvement diminished with increasing reader experience. Among NL/NS residents, confidence increased from 2.60 ± 1.29 to 3.41 ± 1.25. Radiology residents showed a more modest increase from 3.63 ± 1.25 to 3.88 ± 1.15, whereas neuroradiologists exhibited consistently high confidence levels, increasing only minimally from 4.41 ± 0.91 to 4.50 ± 0.96. Analysis of Image Descriptions Across rater groups, imaging descriptions by neuroradiologists achieved the highest ratings for both correctness (median = 4.0, IQR = 0.0; mean = 3.70) and completeness (median = 4.0, IQR = 1.0; mean = 3.38). Radiology residents followed with high scores for correctness (median = 4.0, IQR = 1.0; mean = 3.49) but more moderate scores for completeness (median = 3.0, IQR = 2.0; mean = 2.99). NL/NS residents received the lowest ratings for both correctness (median = 4.0, IQR = 2.0; mean = 3.14) and completeness (median = 2.0, IQR = 1.0; mean = 2.27) ( Figure 6 ). Download figure Open in new tab Figure 6: Expert ratings of reader-generated imaging findings (4-point Likert scale). Diamond symbols indicate mean scores. CLMMs revealed significant positive associations between radiological experience and both prompt correctness (β = 0.113, SE = 0.031, z = 3.714, p < 0.001) and completeness (β = 0.178, SE = 0.056, z = 3.159, p = 0.002). CLMMs revealed significant positive associations between radiological experience and both prompt correctness (β = 0.113, SE = 0.031, z = 3.714, p < 0.001) and completeness (β = 0.178, SE = 0.056, z = 3.159, p = 0.002). Both models exhibited substantial random effects variance for case (correctness: variance = 0.89, SD = 0.94; completeness: variance = 0.93, SD = 0.96) and mixed variance for rater (correctness: variance = 0.18, SD = 0.43; completeness: variance = 1.14, SD = 1.07). Discussion This study evaluated how radiologists with varying levels of experience benefit from LLM assistance in brain MRI differential diagnosis. In summary, we found that absolute performance of LLMs based on reader-generated imaging findings increased with increasing reader experience, while relative diagnostic gains of readers through LLM assistance diminished. Additionally, we validated that the differences in standalone LLM performance were associated with the correctness and completeness of imaging findings, which improved with increasing reader experience, as evaluated by an independent expert. Our findings underline the gap between standalone LLM performance and actual clinical relevance. Even though the LLM achieved its highest performance when provided with input from expert neuroradiologists, this reader group derived only minimal benefit from LLM assistance that might not justify the effort of using such an adjunct tool. In contrast, while the present findings indicate the greatest diagnostic improvement for clinicians from other specialties without formal radiological training, this should not be interpreted as evidence that referring clinicians could use these tools to replace radiologist assessments. First, even with sizeable accuracy gains from LLM assistance, their performance remained markedly below that of neuroradiologists. Second, despite LLM support, NL/NS residents reported significantly lower confidence compared with neuroradiologists, making these assessments unreliable for clinical decision-making. Third, several critical steps in the preceding radiology workflow requiring radiologist expertise - such as protocol selection and the detection of remarkable findings - were presumed in this study, rendering an independent LLM-assisted diagnosis by referring clinicians infeasible. Considering both the practical constraints and the observed diagnostic benefits, radiology residents appear to be the group most likely to derive meaningful advantages from LLM assistance in real-world clinical practice. Crucially, our analysis revealed substantial variation in the diagnostic benefit of LLM assistance across cases. This finding suggests that readers could harness the value of LLM-based decision support more efficiently by utilizing it in selected challenging cases. From a broader perspective, our findings illustrate key determinants of successful human-AI collaboration in the context of generative AI. First, the quality of the content provided by the user strongly determines the quality of the output, as reflected in the standalone LLM performance by reader group. Unlike traditional machine learning systems executing narrowly defined tasks on fixed input data, generative AI systems can perform a much broader range of tasks. Yet, performance is highly contingent on the clarity, precision, and contextual richness of the model prompt. One study reported improvements in diagnostic performance by up to 30% when incorporating lab results in the LLM prompt [ 18 ]. Alarmingly, however, LLMs are also vulnerable to cognitive biases, such as suggestibility bias (prioritizing user agreement over independent reasoning) and framing bias (influence of presentation or wording on decision-making) [ 19 ]. For example, salient but distracting patient history information was shown to impair LLM diagnostic performance [ 20 ]. Second, the capacity of the human user to differentiate between correct and incorrect LLM suggestions is essential. This ability is shaped by the relative degree of trust placed in one’s own judgment versus trust in the model’s outputs, as suggested by prior research on automation bias (overreliance on automated decision-making systems) and algorithmic aversion (excessive skepticism toward algorithmic outputs) [ 21 – 24 ]. Interestingly, a non–domain-specific study on user perceptions of LLM accuracy found that longer explanations increased users’ confidence in the LLM’s responses, even when the actual response accuracy remained unchanged [ 25 ]. In clinical practice, where physicians face enormous workload pressures, cognitive and temporal constraints further limit the users’ capacity to verify the large volumes of content that LLMs can generate [ 26 ]. There are ongoing research efforts to explore methods for quantifying the certainty of LLM outputs, and early findings suggest that token-level probability estimates may provide a more reliable indicator of model uncertainty than self-reported confidence scores, which tend to exhibit overconfidence [ 27 ]. However, the mechanisms by which trust in LLMs is formed, maintained, or undermined within clinical decision support contexts remain largely underexplored, representing a promising avenue for future research. Limitations This study has several limitations. First, this is a single-center study employing a small set of cases in a sub-domain of radiology, limiting the generalizability of findings. Second, we deliberately standardized the interaction of readers with LLMs to avoid the introduction of undesired confounders, but this reduced the realism of the setting. Variations in prompting strategies, choice of models or tools, and the amount of effort devoted to verifying LLM suggestions will likely amplify differences in reader-level benefits. Third, the imaging findings were presented to the models in German, the native language of the readers. Prior studies have shown that LLMs tend to achieve higher performance when prompted in high-resource languages [ 28 ], and providing the inputs in English might have resulted in slightly better model performance. Finally, the models in this study were provided with textual information only. This decision was based on prior evidence indicating only negligible accuracy improvements when including MRI key images as additional input [ 17 ]. Recently, vision-language models (VLM) able to process 3D medical imaging data have been described, but their availability remains limited [ 29 , 30 ]. The diagnostic benefits of human readers from such models remain to be investigated. Conclusion With increasing reader experience, the absolute diagnostic performance of LLMs in brain MRI cases based on reader-generated imaging findings improved, while the relative diagnostic benefit of LLM assistance declined. Our findings call attention to the gap between standalone LLM performance and actual clinical relevance, emphasizing the need to account for human-AI interaction in this context. Data Availability All data produced in the present study are available upon reasonable request to the authors References 1. ↵ Gertz RJ , Bunck AC , Lennartz S , et al. ( 2023 ) GPT-4 for Automated Determination of Radiologic Study and Protocol Based on Radiology Request Forms: A Feasibility Study . Radiology 307 :. doi: 10.1148/RADIOL.230877 OpenUrl CrossRef 2. ↵ Reiner LN , Chelbi M , Fetscher L , et al. ( 2025 ) Automated MRI protocoling in neuroradiology in the era of large language models . Radiologia Medica 1 – 11 . doi: 10.1007/S11547-025-02040-9/TABLES/3 OpenUrl CrossRef 3. ↵ Sun Z , Ong H , Kennedy P , et al. ( 2023 ) Evaluating GPT-4 on Impressions Generation in Radiology Reports . Radiology 307 :. doi: 10.1148/RADIOL.231259 OpenUrl CrossRef 4. ↵ Ziegelmayer S , Marka AW , Lenhart N , et al. ( 2023 ) Evaluation of GPT-4’s Chest X-Ray Impression Generation: A Reader Study on Performance and Perception . J Med Internet Res 25 : e50865 . doi: 10.2196/50865 OpenUrl CrossRef 5. ↵ Wihl J , Rosenkranz E , Schramm S , et al. ( 2025 ) Data extraction from free-text stroke CT reports using GPT-4o and Llama-3.3-70B: the impact of annotation guidelines . Eur Radiol Exp 9 :. doi: 10.1186/s41747-025-00600-2 OpenUrl CrossRef 6. Guellec B Le , Lefèvre A , Geay C , et al. ( 2024 ) Performance of an Open-Source Large Language Model in Extracting Information from Free-Text Radiology Reports . Radiol Artif Intell . doi: 10.1148/RYAI.230364 OpenUrl CrossRef 7. ↵ Meddeb A , Ebert P , Bressem KK , et al. ( 2024 ) Evaluating local open-source large language models for data extraction from unstructured reports on mechanical thrombectomy in patients with ischemic stroke . J Neurointerv Surg jnis-2024-022078 . doi: 10.1136/jnis-2024-022078 OpenUrl Abstract / FREE Full Text 8. ↵ Ueda D , Mitsuyama Y , Takita H , et al. ( 2023 ) ChatGPT’s Diagnostic Performance from Patient History and Imaging Findings on the Diagnosis Please Quizzes . Radiology 308 :. doi: 10.1148/RADIOL.231040 OpenUrl CrossRef 9. Horiuchi D , Tatekawa H , Oura T , et al. ( 2024 ) ChatGPT’s diagnostic performance based on textual vs. visual information compared to radiologists’ diagnostic performance in musculoskeletal radiology . Eur Radiol 1 – 11 . doi: 10.1007/S00330-024-10902-5 OpenUrl CrossRef 10. Sun SH , Huynh K , Cortes G , et al. ( 2024 ) Testing the Ability and Limitations of ChatGPT to Generate Differential Diagnoses from Transcribed Radiologic Findings . Radiology 313 :. doi: 10.1148/RADIOL.232346 , OpenUrl CrossRef 11. ↵ Suh PS , Shim WH , Suh CH , et al. ( 2024 ) Comparing Diagnostic Accuracy of Radiologists versus GPT-4V and Gemini Pro Vision Using Image Inputs from Diagnosis Please Cases . Radiology 312 :. doi: 10.1148/RADIOL.240273 OpenUrl CrossRef PubMed 12. ↵ Kim SH , Wihl J , Schramm S , et al. ( 2025 ) Human-AI collaboration in large language model-assisted brain MRI differential diagnosis: a usability study . Eur Radiol . doi: 10.1007/s00330-025-11484-6 OpenUrl CrossRef 13. ↵ Le Guellec B , Bruge C , Chalhoub N , et al. ( 2025 ) Comparison between multimodal foundation models and radiologists for the diagnosis of challenging neuroradiology cases with text and images . Diagn Interv Imaging . doi: 10.1016/J.DIII.2025.04.006 OpenUrl CrossRef 14. ↵ Siepmann R , Huppertz M , Rastkhiz A , et al. ( 2024 ) The virtual reference radiologist: comprehensive AI assistance for clinical image reading and interpretation . Eur Radiol 1 – 15 . doi: 10.1007/S00330-024-10727-2 OpenUrl CrossRef 15. ↵ Kim SH , Schramm S , Wihl J , et al. ( 2025 ) Boosting LLM-assisted diagnosis: 10-minute LLM tutorial elevates radiology residents’ performance in brain MRI interpretation . Neuroradiology . doi: 10.1007/s00234-025-03664-4 OpenUrl CrossRef 16. ↵ McDuff D , Schaekermann M , Tu T , et al. ( 2025 ) Towards accurate differential diagnosis with large language models . Nature 642 : 451 – 457 . doi: 10.1038/S41586-025-08869-4 OpenUrl CrossRef PubMed 17. ↵ Schramm S , Preis S , Metz M-C , et al. ( 2025 ) Impact of Multimodal Prompt Elements on Diagnostic Performance of GPT-4V in Challenging Brain MRI Cases . Radiology 314 :. doi: 10.1148/radiol.240689 OpenUrl CrossRef 18. ↵ Bhasuran B , Jin Q , Xie Y , et al. ( 2025 ) Preliminary analysis of the impact of lab results on large language model generated differential diagnoses . NPJ Digit Med 8 : 1 – 15 . doi: 10.1038/ OpenUrl CrossRef PubMed 19. ↵ Mahajan A , Obermeyer Z , Daneshjou R , et al. ( 2025 ) Cognitive bias in clinical large language models . NPJ Digit Med 8 : 1 – 4 . doi: 10.1038/S41746-025-01790-0;SUBJMETA OpenUrl CrossRef PubMed 20. ↵ Schmidt HG , Rotgans JI , Mamede S ( 2025 ) Bias Sensitivity in Diagnostic Decision-Making: Comparing ChatGPT with Residents . J Gen Intern Med 40 : 790 – 795 . doi: 10.1007/S11606-024-09177-9/TABLES/2 OpenUrl CrossRef PubMed 21. ↵ Kim SH , Schramm S , Riedel EO , et al. ( 2025 ) Automation bias in AI-assisted detection of cerebral aneurysms on time-of-flight MR angiography . Radiologia Medica . doi: 10.1007/s11547-025-01964-6 OpenUrl CrossRef 22. Dratsch T , Chen X , Mehrizi MR , et al. ( 2023 ) Automation Bias in Mammography: The Impact of Artificial Intelligence BI-RADS Suggestions on Reader Performance . Radiology 307 :. doi: 10.1148/RADIOL.222176 OpenUrl CrossRef 23. Goddard K , Roudsari A , Wyatt JC ( 2012 ) Automation bias: A systematic review of frequency, effect mediators, and mitigators . Journal of the American Medical Informatics Association 19 : 121 – 127 . doi: 10.1136/amiajnl-2011-000089 OpenUrl CrossRef PubMed 24. ↵ Mahmud H , Islam AKMN , Ahmed SI , Smolander K ( 2022 ) What influences algorithmic decision-making? A systematic literature review on algorithm aversion . Technol Forecast Soc Change 175 :. doi: 10.1016/j.techfore.2021.121390 OpenUrl CrossRef 25. ↵ Steyvers M , Tejeda H , Kumar A , et al. ( 2025 ) What large language models know and what people think they know . Nat Mach Intell . doi: 10.1038/s42256-024-00976-7 OpenUrl CrossRef 26. ↵ Ohde JW , Rost LM , Overgaard JD ( 2025 ) The Burden of Reviewing LLM-Generated Content . NEJM AI 2 :. doi: 10.1056/AIp2400979 OpenUrl CrossRef 27. ↵ Bentegeac R , Le Guellec B , Kuchcinski G , et al. ( 2025 ) Token Probabilities to Mitigate Large Language Models Overconfidence in Answering Medical Questions: Quantitative Study . J Med Internet Res 27 : e64348 . doi: 10.2196/64348 OpenUrl CrossRef PubMed 28. ↵ Li Z , Shi Y , Liu Z , et al. ( 2025 ) Language Ranker: A Metric for Quantifying LLM Performance Across High and Low-Resource Languages . Proceedings of the AAAI Conference on Artificial Intelligence 39 : 28186 – 28194 . doi: 10.1609/AAAI.V39I27.35038 OpenUrl CrossRef 29. ↵ Blankemeier L , Cohen JP , Kumar A , et al. ( 2024 ) Merlin: A Vision Language Foundation Model for 3D Computed Tomography . Res Sq . doi: 10.21203/RS.3.RS-4546309/V1 OpenUrl CrossRef 30. ↵ Wu J , Wang Y , Zhong Z , et al. ( 2025 ) Vision-language foundation model for 3D medical imaging . npj Artificial Intelligence 1 :. doi: 10.1038/S44387-025-00015-9 OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted October 28, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following The Expertise Paradox: Who Benefits from LLM-Assisted Brain MRI Differential Diagnosis? Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share The Expertise Paradox: Who Benefits from LLM-Assisted Brain MRI Differential Diagnosis? Severin Schramm , Bastien Le Guellec , Marlene Topka , Mortimer Svec , Paul Backhaus , Viktor Maria Eisenkolb , Evamaria O. Riedel , Mirjam Beyrle , Paul-Sören Platzek , Constanze Ramschütz , Karolin J. Paprottka , Martin Renz , Jannis Bodden , Jan S. Kirschke , Sebastian Ziegelmayer , Felix Busch , Marcus R. Makowski , Lisa Adams , Keno Bressem , Dennis M. Hedderich , Benedikt Wiestler , Su Hwan Kim medRxiv 2025.10.28.25338816; doi: https://doi.org/10.1101/2025.10.28.25338816 Share This Article: Copy Citation Tools The Expertise Paradox: Who Benefits from LLM-Assisted Brain MRI Differential Diagnosis? Severin Schramm , Bastien Le Guellec , Marlene Topka , Mortimer Svec , Paul Backhaus , Viktor Maria Eisenkolb , Evamaria O. Riedel , Mirjam Beyrle , Paul-Sören Platzek , Constanze Ramschütz , Karolin J. Paprottka , Martin Renz , Jannis Bodden , Jan S. Kirschke , Sebastian Ziegelmayer , Felix Busch , Marcus R. Makowski , Lisa Adams , Keno Bressem , Dennis M. Hedderich , Benedikt Wiestler , Su Hwan Kim medRxiv 2025.10.28.25338816; doi: https://doi.org/10.1101/2025.10.28.25338816 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Radiology and Imaging Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (541) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00c101cdfa973a2',t:'MTc3OTYyMzIxOA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.