ChatGPT vs DeepSeek: A Comparative Study of Diagnostic Accuracy and Clinical Reasoning in Rare and Complex Diseases

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Diagnostic errors in rare and complex diseases contribute significantly to morbidity and mortality. The ability of large language models (LLMs) to enhance diagnostic performance in such cases remains uncertain. This study compares the diagnostic accuracy, clinical reasoning quality, and inference efficiency of three ChatGPT variants (o3-mini, o3-mini-high, o1) and DeepSeek-R1 using 30 English-language case reports of rare and complex diseases from 26 specialties across 15 countries, sourced from PubMed and Web of Science Core Collection databases. Cases were selected to avoid overlap with model training data. Each case was processed once by each model, with outputs anonymized and evaluated in a double-blind manner by two board-certified physicians (each with >15 years’ clinical experience) and ChatGPT-4o. Diagnostic accuracy, the primary outcome, ranged between 30.0% and 40.0% with no significant differences observed among models (Cochran’s Q test, P = 0.16). ChatGPT-o1 achieved the highest accuracy (12/30, 40.0%; 95% CI, 24.6%+/-57.7%), followed by ChatGPT-o3-mini and o3-mini-high (each 11/30, 36.7%), and DeepSeek-R1 (9/30, 30.0% for each English and Chinese language inputs). Mean reasoning scores differed significantly (P < 0.05): ChatGPT-o1, 4.08 +/- 0.82; DeepSeek-R1 (English), 3.86 +/- 0.86; ChatGPT-o3-mini, 3.71 +/- 0.90; ChatGPT-o3-mini-high, 3.69 +/- 0.80; DeepSeek-R1 (Chinese), 3.67 +/- 0.84. Inter-evaluator agreement was high (ICC = 0.84; 95% CI, 0.80-0.88). Inference times varied significantly (P < 0.001), with ChatGPT-o3-mini being fastest (7.0 +/- 3.8 s) and DeepSeek-R1 (English) slowest (46.5 +/- 32.5 s). Advanced LLMs demonstrate potential to support diagnosis of rare and complex diseases, with transparent reasoning processes that may aid clinical decision-making and medical education. Further domain-specific refinement and prospective clinical validation are essential for safe and effective integration into clinical practice. Highlights While LLMs showed similar diagnostic accuracy (30-40%) in rare and complex diseases, ChatGPT-o1 significantly excelled in the quality of its clinical reasoning. Inference speeds varied dramatically (7s-47s), highlighting a critical trade-off between model performance and real-world utility. The transparent reasoning of LLMs shows clear promise as a tool to support clinical decision-making and medical education. Safe clinical implementation is dependent on future domain-specific refinement and prospective validation.
Full text 40,686 characters · extracted from preprint-html · click to expand
ChatGPT vs DeepSeek: A Comparative Study of Diagnostic Accuracy and Clinical Reasoning in Rare and Complex Diseases | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search ChatGPT vs DeepSeek: A Comparative Study of Diagnostic Accuracy and Clinical Reasoning in Rare and Complex Diseases View ORCID Profile Jialin Liu , Weiping Cao , Bo Yuan , Wenyi Xie , View ORCID Profile Changyu Wang , Siru Liu doi: https://doi.org/10.1101/2025.08.28.25331796 Jialin Liu 1 Department of Otolaryngology-Head and Neck Surgery, West China Hospital, Sichuan University , Chengdu, China 2 Department of Medical Informatics, West China Hospital, Sichuan University , Chengdu, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jialin Liu For correspondence: DLJL8{at}163.com Weiping Cao 3 Department of Geriatrics, The People’s Hospital of Leshan , Leshan China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Bo Yuan 4 General Practice Medical Center, West China Hospital, Sichuan University , Chengdu, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Wenyi Xie 5 Medical Intensive Care Unit, West China Hospital, Sichuan University , China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Changyu Wang 2 Department of Medical Informatics, West China Hospital, Sichuan University , Chengdu, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Changyu Wang Siru Liu 6 Department of Biomedical Informatics, Vanderbilt University Medical Center , Nashville, TN, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Diagnostic errors in rare and complex diseases contribute significantly to morbidity and mortality. The ability of large language models (LLMs) to enhance diagnostic performance in such cases remains uncertain. This study compares the diagnostic accuracy, clinical reasoning quality, and inference efficiency of three ChatGPT variants (o3-mini, o3-mini-high, o1) and DeepSeek-R1 using 30 English-language case reports of rare and complex diseases from 26 specialties across 15 countries, sourced from PubMed and Web of Science Core Collection databases. Cases were selected to avoid overlap with model training data. Each case was processed once by each model, with outputs anonymized and evaluated in a double-blind manner by two board-certified physicians (each with >15 years’ clinical experience) and ChatGPT-4o. Diagnostic accuracy, the primary outcome, ranged between 30.0% and 40.0% with no significant differences observed among models (Cochran’s Q test, P = 0.16). ChatGPT-o1 achieved the highest accuracy (12/30, 40.0%; 95% CI, 24.6%+/-57.7%), followed by ChatGPT-o3-mini and o3-mini-high (each 11/30, 36.7%), and DeepSeek-R1 (9/30, 30.0% for each English and Chinese language inputs). Mean reasoning scores differed significantly (P < 0.05): ChatGPT-o1, 4.08 +/- 0.82; DeepSeek-R1 (English), 3.86 +/- 0.86; ChatGPT-o3-mini, 3.71 +/- 0.90; ChatGPT-o3-mini-high, 3.69 +/- 0.80; DeepSeek-R1 (Chinese), 3.67 +/- 0.84. Inter-evaluator agreement was high (ICC = 0.84; 95% CI, 0.80-0.88). Inference times varied significantly (P < 0.001), with ChatGPT-o3-mini being fastest (7.0 +/- 3.8 s) and DeepSeek-R1 (English) slowest (46.5 +/- 32.5 s). Advanced LLMs demonstrate potential to support diagnosis of rare and complex diseases, with transparent reasoning processes that may aid clinical decision-making and medical education. Further domain-specific refinement and prospective clinical validation are essential for safe and effective integration into clinical practice. Highlights While LLMs showed similar diagnostic accuracy (30-40%) in rare and complex diseases, ChatGPT-o1 significantly excelled in the quality of its clinical reasoning. Inference speeds varied dramatically (7s-47s), highlighting a critical trade-off between model performance and real-world utility. The transparent reasoning of LLMs shows clear promise as a tool to support clinical decision-making and medical education. Safe clinical implementation is dependent on future domain-specific refinement and prospective validation. Introduction Diagnostic errors present a significant challenge in clinical practice, particularly with regard to rare and complex diseases, and substantially contribute to morbidity and mortality rates. 1 , 2 In U.S. outpatient settings, diagnostic errors affect approximately 5% of adults, 3 while among hospitalized patients, they account for 6% to 17% of adverse events, 4 compromising patient safety and healthcare quality. Annually, these errors result in severe harm to an estimated 795,000 patients in the U.S. (range, 598,000–1,023,000), including approximately 371,000 fatalities and 424,000 cases of permanent disability. 5 Rare and complex diseases are particularly challenging due to atypical presentations, limited diagnostic resources, and the need for specialized expertise, often leading to prolonged diagnostic delays. 6 Clinicians face high workloads, cognitive demands, and increasing burnout rates, which exacerbate cognitive biases and reduce diagnostic accuracy. 7 Advances in artificial intelligence, particularly large language models (LLMs), offer a promising approach to mitigate these challenges by supporting clinical reasoning and decision-making. 8 While LLMs have shown efficacy in medical knowledge assessments and diagnosing common diseases, 8 – 11 their application to rare and complex diseases remains emerging, with limited evidence on their diagnostic accuracy and clinical utility. 12 – 14 This study evaluates the diagnostic performance and reasoning quality of four LLMs—ChatGPT-o3-mini, ChatGPT-o3-mini-high, ChatGPT-o1, and DeepSeek-R1—using real-world cases of rare and complex diseases. By comparing their diagnostic accuracy, reasoning processes, and inference efficiency against clinical expert standards, we aim to elucidate their potential to enhance diagnostic decision-making and inform the development of advanced diagnostic support systems. Methods Case selection We selected 30 clinical case reports of rare and complex diseases from PubMed and Web of Science Core Collection databases, published between January 1, 2024, and February 15, 2025. The search strategy used the keywords “rare disease” OR “complex disease” AND “case report” (title). Records indexed through February 15, 2025, were included, which postdate the December 31, 2024, cutoff for the training data of the large language models (ChatGPT-o3-mini, ChatGPT-o3-mini-high, ChatGPT-o1, DeepSeek-R1). From 806 initial records, duplicates were removed, and two independent reviewers screened titles and abstracts using predefined eligibility criteria. Full texts were assessed for: (1) publication in English in a peer-reviewed journal; (2) comprehensive clinical data (patient history, physical examination, laboratory results, treatment course); (3) detailed narrative descriptions of imaging or multimedia content sufficient for text-based review; (4) representation of diverse clinical specialties (e.g., internal medicine, neurology, oncology, surgery, pediatrics), with the 30 cases collectively spanning at least 26 diverse clinical specialties; (5) balanced geographic representation (Asia, Africa, Europe, North America, South America, Oceania), age groups, sex, and ethnicity; (6) explicit documentation of diagnostic reasoning; and (7) confirmation by a multidisciplinary panel of three board-certified clinical experts. Exclusion criteria included duplicate reports, absence of a definitive diagnosis, or incomplete clinical reasoning descriptions. Disagreements were resolved by consensus or adjudicated by a third expert using a four-point scale (1 = strongly exclude, 4 = strongly include), with cases scoring ≥3 by all experts included. Stratified random sampling ensured balanced representation across specialties and demographics. Models application The study, conducted in February 2025, evaluated four LLMs: ChatGPT-o3-mini, ChatGPT-o3-mini-high, ChatGPT-o1, and DeepSeek-R1. Each model processed the 30 cases once using the prompt: “What is the most likely diagnosis?” Internet access was disabled to prevent external information retrieval. Inputs for ChatGPT models were in English, while DeepSeek-R1 received inputs in English and Chinese (translated by two bilingual experts and validated by ChatGPT-4o for semantic equivalence). Models used default configurations, employing chain-of-thought (CoT) prompting without additional fine-tuning, reflecting typical usage scenarios. Reasoning analysis rating scale To assess the quality of clinical reasoning, five specialists from neurology, geriatrics, otolaryngology, oncology, and nephrology collaboratively developed a 5-point rating scale. The scale was grounded in established principles of diagnostic reasoning and designed to evaluate three key domains: evidentiary completeness, logical coherence, and clinical relevance. 15 – 17 Reasoning was rated on a scale from 1 to 5. 1. Reasoning is entirely unclear, with no logical thread. 2. Poorly structured reasoning and scattered evidence. 3. Broadly complete reasoning but with noticeable logical gaps or insufficient support. 4. Logical, evidence-based reasoning that correctly includes the diagnosis within the differential, albeit with minor omissions.5. Complete, rigorously logical reasoning consistent with medical consensus, yielding a correct and well-justified diagnosis. Evaluation of model reasoning and discrepancy resolution To evaluate LLM-generated clinical reasoning, we conducted a double-blind assessment using a panel of two board-certified physicians (each with over 15 years of experience in different specialties) and ChatGPT-4o. Evaluators underwent standardized rubric-based training, including calibration exercises, to ensure rating consistency. Model outputs were anonymized and randomly ordered. Each evaluator received the clinical case narrative, reference diagnosis with rationale, and a validated 5-point Likert scale rubric assessing logical coherence, evidentiary completeness, and clinical relevance. Materials were provided in Chinese and English, with translations independently verified for accuracy. Cases with a ≥2-point score discrepancy between evaluators were flagged for structured discussion between the physicians. If consensus was not reached, a third blinded clinical expert provided final adjudication. Inter-evaluator reliability was measured using the intraclass correlation coefficient (ICC). Statistical analysis All analyses were performed using Python 3.8 (pandas 1.2.5, SciPy 1.7.3, statsmodels 0.12.2). Diagnostic accuracy was reported as proportions with 95% confidence intervals (CIs). Cochran’s Q test evaluated overall differences in binary outcomes across models, with pairwise comparisons conducted using McNemar’s test and Holm’s Bonferroni correction. To assess the differences in continuous variables among the groups, one-way analysis of variance (ANOVA) was performed. Post hoc pairwise comparisons with Bonferroni correction were conducted to identify significant differences between specific groups. Inter-evaluator reliability was assessed using two-way random-effects ICC, interpreted as: 0.90 excellent. 18 A two-sided P < 0.05 was considered statistically significant. Ethical approval Given the study utilized publicly available and anonymized data previously reviewed by an ethics committee, further ethical approval was waived. Result Clinical cases The 30 selected cases, sourced from 15 countries across six continents and published in 17 peer-reviewed journals, spanned 26 clinical specialties, including oncology, neurology, neurosurgery, infectious diseases, hematology, general surgery, pediatrics, gynecology, immunology, otolaryngology, ophthalmology, allergology, psychosomatic medicine, and oral and maxillofacial surgery. This diverse representation reflects the complexity of real-world diagnostic challenges in rare and complex diseases ( Table 1 ). View this table: View inline View popup Table 1. Geographic origin and publication sources of the 30 rare and difficult cases Diagnostic accuracy and inference time Diagnostic accuracy ranged between 30.0% and 40.0% across models, with no statistically significant differences identified (Cochran’s Q test, P = 0.162). ChatGPT-o1 achieved the highest accuracy (12/30, 40.0%; 95% CI, 24.6%–57.7%), followed by ChatGPT-o3-mini and ChatGPT-o3-mini-high (each 11/30, 36.7%; 95% CI, 21.9%–54.5%), and DeepSeek-R1 (9/30, 30.0%; 95% CI, 16.7%–47.9% for both English and Chinese inputs). No significant differences were observed in pairwise comparisons (McNemar’s test, Holm-corrected P > 0.05) ( Table2 ). View this table: View inline View popup Download powerpoint Table 2. Diagnostic accuracy and inference time of the LLMs Inference times differed significantly (P < 0.001). ChatGPT-o3-mini was fastest (7.0 ± 3.8 s; range 3–17 s), followed by ChatGPT-o3-mini-high (12.4 ± 7.9 s; range 3–41 s) and ChatGPT-o1 (13.4 ± 11.7 s; range 3–49 s). DeepSeek-R1 had the longest inference times (English: 46.5 ± 32.5 s, range 16–161 s; Chinese: 30.7 ± 22.7 s, range 11–137 s), both significantly longer than all other models (p < 0.05). Pairwise comparisons showed no significant difference between ChatGPT-o3-mini-high and ChatGPT-o1 (P = 0.773) ( Figure 1 , 2 and Table 2 ). Download figure Open in new tab Figure 1. Inference time distribution by model (n=30) Download figure Open in new tab Figure 2. Heatmap of holm-corrected P-values for inference time Model Reasoning Evaluation Reasoning scores showed high inter-evaluator agreement (ICC = 0.84; 95% CI, 0.80–0.88). ChatGPT-o1 achieved the highest mean reasoning score (4.08 ± 0.82), significantly outperforming other models (P < 0.05). DeepSeek-R1 (English) scored 3.86 ± 0.86, followed by ChatGPT-o3-mini (3.71 ± 0.90), ChatGPT-o3-mini-high (3.69 ± 0.80), and DeepSeek-R1 (Chinese) (3.67 ± 0.84). No significant differences were observed among the latter three models (P > 0.05) ( Table 3 and Figure 3 ). View this table: View inline View popup Table 3. Inference Scores for ChatGPT and DeepSeek-R1 Models Download figure Open in new tab Figure 3. Mean Inference Scores for ChatGPT and DeepSeek-R1 Models (n=30) Discussion This study evaluated the diagnostic accuracy, reasoning quality, and inference efficiency of four LLMs—ChatGPT-o3-mini, ChatGPT-o3-mini-high, ChatGPT-o1, and DeepSeek-R1—across 30 rare and complex disease cases. Diagnostic accuracy ranged from 30.0% to 40.0%, with no significant differences between models (Cochran’s Q test, P = 0.162), consistent with prior studies. 49 , 50 ChatGPT-o1 demonstrated superior reasoning quality (4.08 ± 0.82), likely due to its advanced CoT optimization 51 , outperforming other models (P 0.05), suggesting that its enhancements do not significantly improve performance in this context. Notably, ChatGPT-o3-mini-high did not demonstrate superior reasoning quality compared to its base model (o3-mini), despite advertised enhancements. 52 DeepSeek-R1’s reasoning performance was better with English inputs (3.86 ± 0.86) than Chinese inputs (3.67 ± 0.84), despite its development for Chinese-speaking users. 53 This may reflect a predominance of English-centric training data or differences in language structure affecting reasoning clarity. Inference times varied significantly, with ChatGPT-o3-mini being fastest (7.0 ± 3.8 s) and DeepSeek-R1 (English) slowest (46.5 ± 32.5 s). The longer inference times for DeepSeek-R1 may be due to its mixture-of-experts architecture, which may involve complex routing and output generation. Inference times varied significantly across models (one-way ANOVA, p < 0.001). ChatGPT-o3-mini was fastest (7.0 ± 3.8 s; range 3–17 s), while DeepSeek-R1 required longer times for English inputs (46.5 ± 32.5 s; range 16–161 s) than Chinese inputs (30.7 ± 22.7 s; range 11–137 s). This discrepancy likely stems from DeepSeek-R1’s mixture-of-experts (MoE) architecture, 50 which increases token counts and computational complexity in CoT decoding, particularly for English prompts. During inference, we observed two operational issues explaining model performance differences. DeepSeek-R1 intermittently returned ‘Server Busy’ errors, whereas ChatGPT remained consistently available, which may reflect backend resource constraints on the DeepSeek platform. Notably, when such errors occurred, DeepSeek-R1 failed to generate outputs. These instances were excluded from reasoning evaluation but did not affect the recorded inference time, as timing was measured only for successful completions. Additionally, under complex prompts, DeepSeek-R1 often produced repetitive or circular reasoning loops—potentially reflecting a reinforcement-learning reward scheme that prioritizes final outcomes over rigorous CoT reasoning. The evaluation methodology employed in this study combined clinical expertise with advanced AI assistance. The hybrid evaluation panel, which consisted of two board-certified physicians from distinct specialties and one large language model evaluator (ChatGPT-4o), facilitated integration of human judgment with AI-driven assessment. This approach enhanced objectivity and reproducibility while maintaining essential clinical standards. A rigorous double-blind evaluation—where model outputs were anonymized and presented in random order—minimized recognition bias. The structured discrepancy resolution protocol, involving calibrated discussions and third-party adjudication, further reinforced scoring consistency. Inter-evaluator reliability was high (ICC = 0.84; 95% CI, 0.80–0.88), substantiating the framework’s robustness. However, inclusion of an LLM evaluator may introduce systematic biases due to model-specific preferences or errors. The limited size of the physician panel may also constrain generalizability, suggesting future studies expand human evaluator participation and include additional calibration rounds to bolster reliability. Limitations and Future Directions Several limitations of this study should be noted. First, our analysis used a relatively small sample of 30 cases. Though diverse, this limits statistical power and external validity. Future work should incorporate larger, more heterogeneous datasets to enhance generalizability. Second, inputs were limited to standardized textual descriptions, which does not leverage the full multimodal capabilities of modern LLMs. More comprehensive assessments could be achieved by integrating direct analysis of medical images or physiological waveforms. Third, the use of an LLM as an evaluator, while innovative, is a relatively unvalidated approach that warrants further investigation. Despite high inter-evaluator agreement (ICC = 0.84), future studies should benchmark this AI-assisted framework against broader panels of human experts. Fourth, while our methodology enhanced objectivity, we did not investigate potential biases related to the reasoning style (e.g., verbose vs. concise) of the LLM outputs and their impact on evaluators. Future research should address these limitations by assessing the real-world utility of these models through prospective clinical trials, for example, by integrating AI-generated differential diagnoses into clinical workflows. Additionally, performance may be enhanced through domain-specific fine-tuning or the use of retrieval-augmented generation. Finally, to optimize usability, user interfaces should be designed with cognitive load theory in mind, perhaps offering a dual-mode presentation with both rapid answers and detailed reasoning traces. Conclusion This study demonstrates that advanced LLMs exhibit measurable capabilities in diagnosing rare and complex diseases. Their transparent reasoning pathways offer valuable clinical insights, fostering physician trust essential for real-world adoption. ChatGPT-o1 showed slightly stronger overall performance, but its proprietary nature limits on-premises deployment. Conversely, DeepSeek-R1’s open-source framework enables local customization and enhanced data control. Selecting an AI model for clinical practice requires balancing diagnostic accuracy, reasoning transparency, data-privacy compliance, inference latency, operational costs, and deployment feasibility. Data Availability All data produced in the present study are available upon reasonable request to the authors Reference 1. ↵ Yang D , Fineberg HV , Cosby K. Diagnostic Excellence . JAMA . 2021 ; 326 ( 19 ): 1905 – 1906 . doi: 10.1001/jama.2021.19493 OpenUrl CrossRef PubMed 2. ↵ Stoller JK . The Challenge of Rare Diseases . Chest . 2018 ; 153 ( 6 ): 1309 – 1314 . doi: 10.1016/j.chest.2017.12.018 OpenUrl CrossRef PubMed 3. ↵ Singh H , Meyer AND , Thomas EJ . The frequency of diagnostic errors in outpatient care: estimations from three large observational studies involving US adult populations . BMJ Qual Saf . 2014 ; 23 ( 9 ): 727 – 731 . doi: 10.1136/bmjqs-2013-002627 OpenUrl Abstract / FREE Full Text 4. ↵ Hall KK , Shoemaker-Hunt S , Hoffman L , et al. Making Healthcare Safer III: A Critical Analysis of Existing and Emerging Patient Safety Practices . Agency for Healthcare Research and Quality (US) ; 2020 . Accessed May 3, 2025 . http://www.ncbi.nlm.nih.gov/books/NBK555526/ 5. ↵ Newman-Toker DE , Nassery N , Schaffer AC , et al. Burden of Serious Harms from Diagnostic Error in the United States . BMJ Qual Saf . 2024 ; 33 ( 2 ): 109 – 120 . doi: 10.1136/bmjqs-2021-014130 OpenUrl Abstract / FREE Full Text 6. ↵ Bordini BJ , Stephany A , Kliegman R. Overcoming Diagnostic Errors in Medical Practice . The Journal of Pediatrics . 2017 ; 185 : 19 – 25 .e1. doi: 10.1016/j.jpeds.2017.02.065 OpenUrl CrossRef PubMed 7. ↵ Hodkinson A , Zhou , A , Johnson J , et al. Associations of physician burnout with career engagement and quality of patient care: systematic review and meta-analysis . BMJ . 2022 ; 378 : e070442 . doi: 10.1136/bmj-2022-070442 OpenUrl Abstract / FREE Full Text 8. ↵ Liu S , Wright AP , Patterson BL , et al. Using AI-generated suggestions from ChatGPT to optimize clinical decision support . J Am Med Inform Assoc . 2023 ; 30 ( 7 ): 1237 – 1245 . doi: 10.1093/jamia/ocad072 OpenUrl CrossRef PubMed 9. Liu J , Wang C , Liu S. Utility of ChatGPT in Clinical Practice . J Med Internet Res . 2023 ; 25 : e48568 . doi: 10.2196/48568 OpenUrl CrossRef PubMed 10. Takita H , Kabata D , Walston SL , et al. A systematic review and meta-analysis of diagnostic performance comparison between generative AI and physicians . NPJ Digit Med . 2025 Mar 22; 8 ( 1 ): 175 . doi: 10.1038/s41746-025-01543-z . OpenUrl CrossRef PubMed 11. ↵ Hager P , Jungmann F , Holland R , et al. Evaluation and mitigation of the limitations of large language models in clinical decision-making . Nat Med . 2024 ; 30 ( 9 ): 2613 – 2622 . doi: 10.1038/s41591-024-03097-1 OpenUrl CrossRef PubMed 12. ↵ Alessandro L , Bianciotti N , Salama L , et al. Artificial Intelligence-Based Virtual Assistant for the Diagnostic Approach of Chronic Ataxias . Mov Disord . Published online March 22, 2025 . doi: 10.1002/mds.30168 OpenUrl CrossRef 13. Young CC , Enichen E , Rivera C , et al. Diagnostic Accuracy of a Custom Large Language Model on Rare Pediatric Disease Case Reports . Am J Med Genet A . 2025 ; 197 ( 2 ): e63878 . doi: 10.1002/ajmg.a.63878 OpenUrl CrossRef PubMed 14. ↵ Tefera L , Rosenzveig A , Rajendran J , et al. Large language models in rare disease: accuracy in addressing fibromuscular dysplasia questions . Vasa . Published online January 9, 2025 . doi: 10.1024/0301-1526/a001175 OpenUrl CrossRef 15. ↵ Goh E , Gallo R , Strong E , et al. Large Language Model Influence on Management Reasoning: A Randomized Controlled Trial . Published online August 7, 2024 :2024.08.05.24311485. doi: 10.1101/2024.08.05.24311485 OpenUrl Abstract / FREE Full Text 16. Fürstenberg S , Helm T , Prediger S , Kadmon M , Berberat PO , Harendza S. Assessing clinical reasoning in undergraduate medical students during history taking with an empirically derived scale for clinical reasoning indicators . BMC Med Educ . 2020 ; 20 ( 1 ): 368 . doi: 10.1186/s12909-020-02260-9 OpenUrl CrossRef PubMed 17. ↵ Cohen A , Sur M , Weisse M , et al. Teaching Diagnostic Reasoning to Faculty Using an Assessment for Learning Tool: Training the Trainer . MedEdPORTAL . 2020 ; 16 : 10938 . doi: 10.15766/mep_2374-8265.10938 OpenUrl CrossRef PubMed 18. ↵ Thayaparan AJ , Mahdi E. The Patient Satisfaction Questionnaire Short Form (PSQ-18) as an adaptable, reliable, and validated tool for use in various settings . Med Educ Online . 2013 ; 18 :10.3402/meo.v18i0.21747. doi: 10.3402/meo.v18i0.21747 OpenUrl CrossRef PubMed 19. Luo Q , Wu S. Orbital mesenchymal chondrosarcoma: a case report . J Int Med Res . 2025 ; 53 ( 1 ): 3000605241311443 . doi: 10.1177/03000605241311443 OpenUrl CrossRef PubMed 20. Yan C , Zhang CJ , Wei JB , Liang HW , Qu S. A case report of comprehensive treatment for primary intraspinal carcinosarcoma . Front Oncol . 2024 ; 14 : 1479193 . doi: 10.3389/fonc.2024.1479193 OpenUrl CrossRef PubMed 21. Qin Z , Li L , Jing T , Wang C. Case report: A rare case of chondrosarcoma-like malignant giant cell tumor in adolescent rib: diagnostic challenges and treatment . Front Oncol . 2024 ; 14 : 1523104 . doi: 10.3389/fonc.2024.1523104 OpenUrl CrossRef PubMed 22. Kim DH , Lee Y. Hemicrania continua with rhinosinusitis: a case report . Korean J Fam Med . 2025 ; 46 ( 1 ): 48 – 51 . doi: 10.4082/kjfm.24.0178 OpenUrl CrossRef PubMed 23. Yang X , Xu C , Richard SA , et al. Case report: Intradural-extramedullary cervical spine clear cell meningioma mimicking a schwannoma in a child . Front Oncol . 2024 ; 14 : 1505141 . doi: 10.3389/fonc.2024.1505141 OpenUrl CrossRef PubMed 24. Chang J , Jin Y , Cui C , Cheng H. Uterine Adenomyosarcoma Complicated by Uterine Prolapse and Necrosis: A Case Report . Int Med Case Rep J . 2025 ; 18 : 15 – 22 . doi: 10.2147/IMCRJ.S489194 OpenUrl CrossRef PubMed 25. Indriasari V , Balia KZ , Usman HA . Anterior urethral hamartoma in a female infant with anorectal malformation: A rare case report and literature review . Int J Surg Case Rep . 2025 ; 128 : 110989 . doi: 10.1016/j.ijscr.2025.110989 OpenUrl CrossRef PubMed 26. Abdul-Hafez HA , Abu-Alrub RA , Barakat MA , Meri AZ , Rass HA . An unusual presentation of primary synovial sarcoma of the ethmoid sinus in a 54-year-old woman: A case report and literature review . Int J Surg Case Rep . 2025 ; 126 : 110784 . doi: 10.1016/j.ijscr.2024.110784 OpenUrl CrossRef PubMed 27. Sandakly N , El Koubayati G , Sarkis J , Haddad F. Pediatric Red Ear Syndrome Misdiagnosed as Relapsing Polychondritis: A Case Report and Review of Literature . Case Rep Pediatr . 2025 ; 2025 : 6464822 . doi: 10.1155/crpe/6464822 OpenUrl CrossRef PubMed 28. Shahoud M , Abdin D , Ismail A , Ismail F , Azzawi T. Carcinoid tumor in a 10-year-old boy challenges in diagnosis and management: A rare case report . Int J Surg Case Rep . 2025 ; 126 : 110811 . doi: 10.1016/j.ijscr.2024.110811 OpenUrl CrossRef PubMed 29. Sivasubramanian D , Aravind S , Sanil S , Senthilkumar V , Rathika V. Isolated fallopian tube torsion with hydrosalpinx in a 14 year old girl: A case report . Int J Surg Case Rep . 2025 ; 127 : 110880 . doi: 10.1016/j.ijscr.2025.110880 OpenUrl CrossRef PubMed 30. Tsai TC , Su CM . A case of hepatic sarcomatoid cholangiocarcinoma with diaphragmatic and right lower lobe lung invasion . J Surg Case Rep . 2025 ; 2025 ( 2 ): rjaf047 . doi: 10.1093/jscr/rjaf047 OpenUrl CrossRef PubMed 31. Kudsi M , Tarcha R , Khalayli N , Rabah N , Rabah K , Alghawe FA . Progression to end-stage renal disease due to IgG4-related nephritis: a case report . Oxf Med Case Reports . 2025 ; 2025 ( 1 ): omae179 . doi: 10.1093/omcr/omae179 OpenUrl CrossRef 32. Ahmad Z , Jehanzeb H , Hussain SN , Umar M , Saleem H. A synchronous occurrence of breast cancer and pleural mesothelioma: a case report . J Med Case Rep . 2025 ; 19 ( 1 ): 25 . doi: 10.1186/s13256-024-04949-7 OpenUrl CrossRef PubMed 33. Majidi F , Shabbak A , Nazarizadeh S , Madady A. Concomitant pheochromocytoma and hyperaldosteronism in a 47-year-old man: a case report . J Med Case Rep . 2025 ; 19 ( 1 ): 20 . doi: 10.1186/s13256-025-05026-3 OpenUrl CrossRef PubMed 34. Feleke AA , Mekonnen DC , Zeneber MB , Zemariam MA , Workneh GA , Amare AG . Peritoneal hydatidosis secondary to ruptured hepatic hydatid cyst-a rare presentation: a case report . J Surg Case Rep . 2025 ; 2025 ( 2 ): rjaf005 . doi: 10.1093/jscr/rjaf005 OpenUrl CrossRef PubMed 35. Reta BK , Beker AM , Hagos HH , Weldegebriel MH , Kidanu GT , Zeray MA . A rare case of primary mucinous cystadenoma of spleen: A case report . Int J Surg Case Rep . 2025 ; 126 : 110782 . doi: 10.1016/j.ijscr.2024.110782 OpenUrl CrossRef PubMed 36. Geremew TT , Zewdie WJ , Nisiro AM , Engida GG , Tesgera TG . A rare case of primary unilateral conjunctival small lymphocytic lymphoma: A case report . Int J Surg Case Rep . 2025 ; 126 : 110812 . doi: 10.1016/j.ijscr.2024.110812 OpenUrl CrossRef PubMed 37. Lugata J , Rapheal A , Makower L , Mchome B , Batchu N. Management challenges of postpartum rhinocerebral mucormycosis following spontaneous vaginal delivery in a resource-constraint setting: A case report and review of literature . Int J Surg Case Rep . 2025 ; 126 : 110834 . doi: 10.1016/j.ijscr.2025.110834 OpenUrl CrossRef PubMed 38. Rahmouni E , Romdhane RB , Boukhris S , Mansouri H , Henchiri H , Achouri L. Rosai-Dorfman disease: A rare presentation as an isolated axillary lymphadenopathy, a case report and literature review . Int J Surg Case Rep . 2025 ; 126 : 110762 . doi: 10.1016/j.ijscr.2024.110762 OpenUrl CrossRef PubMed 39. Leite-Almeida L , Silva D , Dias M , Bordalo D , Jacob S. Recurrent Hand Oedema and Abdominal Pain . J Paediatr Child Health . 2025 ; 61 ( 3 ): 516 – 517 . doi: 10.1111/jpc.16783 OpenUrl CrossRef PubMed 40. Leponce S , Buxant F , Noël JC . Primary retroperitoneal mucinous carcinoma with BRAF, KIT, NF2, and AR mutations: A case report and review of the literature . Case Rep Womens Health . 2025 ; 45 : e00681 . doi: 10.1016/j.crwh.2025.e00681 OpenUrl CrossRef 41. Lafuente-Ibáñez de Mendoza I , Aguirre-Echevarria P , Silva-Soria TM , Aisa FJV , de Larrinoa AF , Aguirre-Urizar JM . Peri-Implant Epstein-Barr Virus (+) Mucocutaneous Ulcer in an Immunocompetent Patient: Case Report and Review of the Literature . Clin Implant Dent Relat Res . 2025 ; 27 ( 1 ): e13440 . doi: 10.1111/cid.13440 OpenUrl CrossRef PubMed 42. Vieira Afonso JFF , Santos MM , Vieira J , Heeren L , Rodrigues AF . Pseudomyxoma Peritonei: A Case Report of a Patient With Unexplained Granulomas . Cureus . 2025 ; 17 ( 1 ): e77173 . doi: 10.7759/cureus.77173 OpenUrl CrossRef 43. Ojardias E , Leman M , Lafaie L , Oriol P , Calmels P , Celarier T. Singular case report of familial hypocalciuric hypercalcemia: a rare diagnosis of hypercalcemia in the older people . Aging Male . 2025 ; 28 ( 1 ): 2436877 . doi: 10.1080/13685538.2024.2436877 OpenUrl CrossRef PubMed 44. Paulino Ferreira L , Alves J , Marta J , Bonifácio GV , Militão A. Munchausen Syndrome Presented as Guillain-Barré Syndrome: A Case Report and Literature Review . Cureus . 2025 ; 17 ( 1 ): e77057 . doi: 10.7759/cureus.77057 OpenUrl CrossRef PubMed 45. Vindel Valle LM , López Alfaro MA. First Reported Case of atypical Cogan’s Syndrome in Central America . Arch Soc Esp Oftalmol (Engl Ed) . 2025 ; 100 ( 1 ): 42 – 45 . doi: 10.1016/j.oftale.2024.08.005 OpenUrl CrossRef PubMed 46. Ekin U , Hazari A , Alyassin N , Alcantara A , Azzam MH , Ismail M. Successful Management of Pseudo-Ludwig Angina from Supratherapeutic Warfarin Use: A Case Report . Clin Pract Cases Emerg Med . 2025 ; 9 ( 1 ): 90 – 94 . doi: 10.5811/cpcem.20386 OpenUrl CrossRef PubMed 47. Castillo M , Hanuch F , Rauch G , Avendaño P , Cuevas O. Lingual artery thrombosis as a presentation of infective endocarditis in a pregnant patient: a case report . Eur Heart J Case Rep . 2025 ; 9 ( 1 ): ytae550 . doi: 10.1093/ehjcr/ytae550 OpenUrl CrossRef 48. Lorger S , Jackson S , Mukhtiar A , Santos L , Gassner P. A rare case of lipomatous ganglioneuroma of the adrenal gland . Urol Case Rep . 2025 ; 59 : 102933 . doi: 10.1016/j.eucr.2025.102933 OpenUrl CrossRef PubMed 49. ↵ Kanjee Z , Crowe B , Rodman A. Accuracy of a Generative Artificial Intelligence Model in a Complex Diagnostic Challenge . JAMA . 2023 ; 330 ( 1 ): 78 – 80 . doi: 10.1001/jama.2023.8288 OpenUrl CrossRef PubMed 50. ↵ DeepSeek-AI , Guo D , Yang D , et al. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning . Published online January 22, 2025 . doi: 10.48550/arXiv.2501.12948 OpenUrl CrossRef 51. ↵ Learning to reason with LLMs . Accessed May 3, 2025 . https://openai.com/index/learning-to-reason-with-llms/ 52. ↵ Comparing OpenAI’s o3 mini vs o3 mini high vs o1 pro . Which is the Best for You? Accessed May 3, 2025 . https://writingmate.ai/blog/openai-o3-mini-high-vs-o1-pro 53. ↵ DeepSeek . Accessed May 3, 2025 . https://www.deepseek.com/ View the discussion thread. Back to top Previous Next Posted August 28, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following ChatGPT vs DeepSeek: A Comparative Study of Diagnostic Accuracy and Clinical Reasoning in Rare and Complex Diseases Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share ChatGPT vs DeepSeek: A Comparative Study of Diagnostic Accuracy and Clinical Reasoning in Rare and Complex Diseases Jialin Liu , Weiping Cao , Bo Yuan , Wenyi Xie , Changyu Wang , Siru Liu medRxiv 2025.08.28.25331796; doi: https://doi.org/10.1101/2025.08.28.25331796 Share This Article: Copy Citation Tools ChatGPT vs DeepSeek: A Comparative Study of Diagnostic Accuracy and Clinical Reasoning in Rare and Complex Diseases Jialin Liu , Weiping Cao , Bo Yuan , Wenyi Xie , Changyu Wang , Siru Liu medRxiv 2025.08.28.25331796; doi: https://doi.org/10.1101/2025.08.28.25331796 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (570) Allergy and Immunology (863) Anesthesia (301) Cardiovascular Medicine (4442) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1511) Epidemiology (15231) Forensic Medicine (30) Gastroenterology (1126) Genetic and Genomic Medicine (6610) Geriatric Medicine (668) Health Economics (998) Health Informatics (4542) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15924) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6608) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1146) Occupational and Environmental Health (957) Oncology (3338) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1693) Pharmacology and Therapeutics (692) Primary Care Research (712) Psychiatry and Clinical Psychology (5450) Public and Global Health (9240) Radiology and Imaging (2203) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (596) Sexual and Reproductive Health (714) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a020861b084c0db4',t:'MTc3OTgzNzc2Ng=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00