Full text
24,649 characters
· extracted from
preprint-html
· click to expand
A Novel Framework for Evaluating the Clinical Reasoning Process of Large Language Models: A Comparative Study in Nephrology | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search A Novel Framework for Evaluating the Clinical Reasoning Process of Large Language Models: A Comparative Study in Nephrology Yuichiro Yano , Hiroaki Kakizaki , Hajime Nagasu , Seiji Kishi , Takeo Koshida , Yoshihito Nihei , Akira Hirano , Masaomi Nangaku , Hirotake Mori , Toshio Naito , Mizuki Ohashi , Shoichi Maruyama , Isao Matsui , Yoshitaka Isaka , Yusuke Suzuki , Naoki Kashihara doi: https://doi.org/10.1101/2025.09.04.25334460 Yuichiro Yano 1 Department of General Medicine, Juntendo University Faculty of Medicine , Tokyo, Japan 2 Artificial Intelligence Incubation Farm, Juntendo University Faculty of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: y.yano{at}juntendo.ac.jp yano.yuichiro{at}jichi.ac.jp Hiroaki Kakizaki 3 PeopleMedia, Inc , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hajime Nagasu 4 Department of Nephrology and Hypertension, Kawasaki Medical School , Kurashiki, Okayama, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Seiji Kishi 4 Department of Nephrology and Hypertension, Kawasaki Medical School , Kurashiki, Okayama, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Takeo Koshida 5 Department of Nephrology, Juntendo University Faculty of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yoshihito Nihei 5 Department of Nephrology, Juntendo University Faculty of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Akira Hirano 4 Department of Nephrology and Hypertension, Kawasaki Medical School , Kurashiki, Okayama, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Masaomi Nangaku 6 Division of Nephrology and Endocrinology, The University of Tokyo Graduate School of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hirotake Mori 1 Department of General Medicine, Juntendo University Faculty of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Toshio Naito 1 Department of General Medicine, Juntendo University Faculty of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mizuki Ohashi 1 Department of General Medicine, Juntendo University Faculty of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Shoichi Maruyama 7 Department of Nephrology, Nagoya University Graduate School of Medicine , Aichi, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Isao Matsui 8 Department of Nephrology, Osaka University Graduate School of Medicine , Suita, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yoshitaka Isaka 8 Department of Nephrology, Osaka University Graduate School of Medicine , Suita, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yusuke Suzuki 5 Department of Nephrology, Juntendo University Faculty of Medicine , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Naoki Kashihara 4 Department of Nephrology and Hypertension, Kawasaki Medical School , Kurashiki, Okayama, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Although interest in the application of large language models (LLMs) in medicine is growing, accuracy evaluations have largely relied on static knowledge tests. However, discussions on clinical reasoning, the process most critical to real-world practice, remain limited. In this study, we propose a novel framework to evaluate not the final diagnosis generated by AI, but the reasoning process itself. This study proposes a novel framework that systematically evaluates the capabilities of LLMs (OpenAI GPT-o3, Gemini 2.5 Pro, DeepSeek-R1, Llxsama4-Marveric) by deconstructing the clinical reasoning process into discrete cognitive steps. We focused on nephrology cases, which often involve multiple organ systems and diverse pathologies, thus requiring a high level of reasoning. The four nephrologists independently evaluated the outputs. Our evaluation of four leading LLMs revealed that while Gemini 2.5 Pro demonstrated the best overall performance, all models exhibited common weaknesses in advanced, synthetic tasks such as “formulating differential diagnoses with rationale” and “treatment planning,” particularly in dynamically changing clinical scenarios. Furthermore, a notable finding of our research is that the highest-performing model was not the most computationally intensive, demonstrating that reasoning quality and computational efficiency are not in a simple trade-off. In conclusion, our step-by-step evaluation method is an effective approach for identifying the specific strengths and weaknesses in an LLM’s clinical reasoning. The weaknesses identified, particularly in formulating a differential diagnosis with a clear rationale and developing comprehensive treatment plans for dynamic scenarios, should become a primary target for future model development and for the creation of support system. Introduction While the application of large language models (LLMs) in medicine is gaining significant attention, their performance evaluation has predominantly relied on static knowledge tests, such as medical licensing exams. These methods fail to adequately reflect the dynamic reasoning process of actual clinical practice, where physicians iteratively form and revise hypotheses based on ongoing patient interactions and test results. 1 , 2 Building on these advancements, our study proposes a novel evaluation framework where specialists scrutinize the “clinical reasoning process itself,” 3 not just the final output of an AI. We segment each clinical case into three distinct stages: patient history, test results, and subsequent clinical course. At each stage, the validity of the AI’s diagnostic reasoning is assessed in a step-by-step manner by expert physicians. Furthermore, we conduct a comparative analysis of leading LLMs with different architectures and levels of openness (OpenAI GPT-o3, Gemini 2.5 Pro, DeepSeek-R1, Llama4-Marveric) to investigate how these structural differences impact the quality of clinical reasoning. By also recording and analyzing efficiency metrics such as response time and token count, we aim to examine their correlation with reasoning ability and assess each model’s effectiveness from a practical standpoint, considering the balance between accuracy and computational resources. To ground our evaluation in a domain that is both complex and structured, this study specifically focused on clinical reasoning within the field of nephrology. 4 This specialty is chosen because kidney diseases are often interconnected with multiple organ systems and present with diverse pathologies, demanding a high degree of clinical reasoning. Despite this complexity, the field has well-systematized diagnostic algorithms and clinical patterns, making it an ideal testbed for validating the reasoning process. Methods A detailed description of the methods can be found in the separate supplementary file. Briefly, four nephrologists used the Delphi method to select ten cases that met the inclusion criteria from over one hundred case reports. As permission could not be obtained from the journal for one of the cases, a total of nine cases were included in the final analysis. For each case, nine questions regarding clinical reasoning were established. The evaluation of the LLM outputs was conducted on July 20, 2025 (JST). Each question was scored on a 3-point scale (0 = Incorrect, 1 = Reasonable but not the best, 2 = Correct). The four nephrologists independently evaluated the outputs in a blinded manner, without knowing the model names. The results were then aggregated and analyzed by a single, blinded researcher. Results Overall Diagnostic Reasoning Performance Across all clinical cases and reasoning questions, a statistically significant difference in overall performance was observed among the four LLMs ( Figure 1A ). Gemini 2.5 Pro achieved the highest average score (7.57), followed by OpenAI’s O3 (7.39), DeepSeek-R1 (7.13), and Llama 4 Maverick (6.23). Download figure Open in new tab Figure 1A. Scores by model for each case and question Scores for each case and question are displayed by model as box-and-whisker plots. Overall differences were assessed with the Kruskal – Wallis test; when significant, pairwise comparisons between models were performed with Holm’s correction for multiple testing. Performance on Specific Reasoning Tasks Based on the performance across all models (( Figure 1B ), the tasks that proved most challenging were Question 2 (“Differential diagnoses and rationale”) and Question 7 (“Treatment Planning”), which received the lowest mean scores of 6.56 and 6.58, respectively. In contrast, the models performed best on Question 1 (“Summarizing medical problems”) and Question 6 (“Reassessment of the differential diagnoses”), which both achieved the highest mean score of 7.50. Download figure Open in new tab Figure 1B. Distribution of Scores by Clinical Reasoning Question (All Models Combined) Box-and-whisker plots show the distribution of scores for each of the nine clinical reasoning questions, aggregating the results from all four LLMs. The box represents the interquartile range (IQR), the line inside the box indicates the median, and the whiskers show the range of the data. Triangles mark the mean score for each question. Questions (Q1–Q9): Q1, summary of the medical problem; Q2, differential diagnoses and rationale; Q3, necessary physical examinations and rationale; Q4, plan for investigations/tests; Q5, interpretation of test results; Q6, reassessment of the differential diagnoses; Q7, treatment planning; Q8, evaluation of treatment; Q9, management in case of clinical worsening. Figure 1C shows heatmap of average scores by model for each clinical reasoning question. Gemini 2.5 Pro demonstrated superior or competitive performance, particularly in complex tasks such as reconsidering the differential diagnosis (Q6, score: 7.89), interpreting test results (Q5, score: 8.00), and formulating plans for worsening conditions (Q9, score: 8.00). Download figure Open in new tab Figure 1C. Heatmap of average scores by model for each clinical reasoning question The values are presented as a heatmap, where red outlines highlight the highest-scoring model for each question. Statistical significance was assessed using the Kruskal–Wallis test, followed by pairwise comparisons with Holm’s correction for multiple testing. An asterisk ( * ) indicates a statistically significant difference in performance compared to Llama 4 Maverick Download figure Open in new tab Figure 1D. Comparison of Computational Efficiency by Model: Time and Token Usage Box-and-whisker plots compare the distribution of (A) time required and (B) tokens consumed per case for each of the four LLMs. The box represents the interquartile range (IQR), the line inside the box is the median, and the whiskers show the range of the data. Triangles mark the mean values for each model. Statistical differences were assessed using the Kruskal–Wallis test, followed by pairwise comparisons with Holm’ s correction. Brackets at the top of the plots indicate statistically significant differences between model pairs Efficiency and Resource Usage While Gemini 2.5 Pro demonstrated high efficiency with a mean response time of 133.2 sec, the greatest computational cost was borne by DeepSeek-R1, which was the slowest (mean time: 260.3 sec) and consumed the most tokens (16,687.2). In contrast, Llama 4 Maverick had the shortest response time and the lowest token usage. Discussion A key novelty of this study lies in its introduction of a framework that deconstructs the clinical reasoning process into discrete cognitive steps and evaluates an AI’s capability at each stage. Previous landmark studies, such as those on Google’s AMIE and Microsoft’s MAI-DxO, 1 , 2 , 5 made significant progress by shifting evaluation away from static tasks toward more dynamic processes like conversational dialogue and sequential information gathering. However, these studies did not pinpoint which specific cognitive tasks within the broader reasoning process represented a model’s weak points. Our step-by-step evaluation approach revealed specific weaknesses in the models’ clinical reasoning. Notably, the most challenging tasks were Question 2 (“Differential diagnoses and rationale”) and Question 7 (“Treatment Planning”). This suggests that while AIs may excel at discrete tasks like summarizing information and interpreting test results, a capability gap remains in more advanced, synthetic skills like planning optimal interventions. 6 This inadequacy was particularly evident in cases requiring dynamic adjustments due to a dramatic change in the clinical condition. For example, in a case involving the sudden onset of nocardiosis during treatment for nephrotic syndrome, many of our expert evaluators indicated that the treatment plans proposed by the models were insufficient. Our research reveals a nuanced relationship between reasoning quality and computational efficiency, challenging the assumption of a simple trade-off. The highest-performing model, Gemini 2.5 Pro, was not the most resource-intensive; in fact, the greatest computational cost was borne by the lower-performing DeepSeek-R1, which was the slowest and consumed the most tokens. This finding is critical for clinical implementation, as it proves that superior reasoning quality does not necessarily demand the highest computational cost, allowing for the selection of AI tools that balance high accuracy with the practical demands of a clinical workflow. In conclusion, our step-by-step evaluation method is an effective approach for identifying the specific strengths and weaknesses in an LLM’s clinical reasoning. The weaknesses identified, particularly in formulating a differential diagnosis with a clear rationale and developing comprehensive treatment plans for dynamic scenarios, should become a primary target for future model development and for the creation of support system. Data Availability The datasets generated or analyzed during this study are available from the corresponding author on reasonable request. Conflicts of Interest None declared. Author contribution YY conceptualized the study and developed the methodology. HN, SK, TK, and YN curated the data. HK and MO conducted the formal analysis. MN, SM, IM, YI, YS, and NK supervised the study. YY drafted the manuscript, and all authors reviewed, edited, and approved the final version. Supplementary Material (PDF) Supplementary methods Supplementary Table 1. Prompts and Evaluation Criteria for the Phased Assessment of Clinical Reasoning Supplementary Table 2. Responses from Four Large Language Models to Clinical Reasoning Questions Across Nine Cases Acknowledgments This research was partially funded by the Advanced Medical Personnel Training Program (principal investigator: TN) and was supported by the Ministry of Education, Culture, Sports, Science, and Technology. References 1. ↵ McDuff D , Schaekermann M , Tu T , et al. Towards accurate differential diagnosis with large language models . Nature . 2025 ; 642 ( 8067 ): 451 – 457 . OpenUrl PubMed 2. ↵ Tu T , Schaekermann M , Palepu A , et al. Towards conversational diagnostic artificial intelligence . Nature . 2025 ; 642 ( 8067 ): 442 – 450 . OpenUrl PubMed 3. ↵ Bowen JL . Educational strategies to promote clinical diagnostic reasoning . N Engl J Med . 2006 ; 355 ( 21 ): 2217 – 2225 . OpenUrl CrossRef PubMed Web of Science 4. ↵ Boyle SM , Martindale J , Parsons AS , et al. Development and Validation of a Formative Assessment Tool for Nephrology Fellows’ Clinical Reasoning . Clin J Am Soc Nephrol . 2024 ; 19 ( 1 ): 26 – 34 . OpenUrl PubMed 5. ↵ Nori H , Daswani M , Kelly C , et al. Sequential Diagnosis with Language Models . arXiv . 2025 . https://arxiv.org/abs/2506.22405 6. ↵ Hager P , Jungmann F , Holland R , et al. Evaluation and mitigation of the limitations of large language models in clinical decision-making . Nat Med . 2024 ; 30 ( 9 ): 2613 – 2622 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted September 07, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A Novel Framework for Evaluating the Clinical Reasoning Process of Large Language Models: A Comparative Study in Nephrology Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A Novel Framework for Evaluating the Clinical Reasoning Process of Large Language Models: A Comparative Study in Nephrology Yuichiro Yano , Hiroaki Kakizaki , Hajime Nagasu , Seiji Kishi , Takeo Koshida , Yoshihito Nihei , Akira Hirano , Masaomi Nangaku , Hirotake Mori , Toshio Naito , Mizuki Ohashi , Shoichi Maruyama , Isao Matsui , Yoshitaka Isaka , Yusuke Suzuki , Naoki Kashihara medRxiv 2025.09.04.25334460; doi: https://doi.org/10.1101/2025.09.04.25334460 Share This Article: Copy Citation Tools A Novel Framework for Evaluating the Clinical Reasoning Process of Large Language Models: A Comparative Study in Nephrology Yuichiro Yano , Hiroaki Kakizaki , Hajime Nagasu , Seiji Kishi , Takeo Koshida , Yoshihito Nihei , Akira Hirano , Masaomi Nangaku , Hirotake Mori , Toshio Naito , Mizuki Ohashi , Shoichi Maruyama , Isao Matsui , Yoshitaka Isaka , Yusuke Suzuki , Naoki Kashihara medRxiv 2025.09.04.25334460; doi: https://doi.org/10.1101/2025.09.04.25334460 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Nephrology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4538) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (541) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3333) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00dd82f98e6ad07',t:'MTc3OTY0MTg5OQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.