Evaluating Large Language Model Diagnostic Performance on JAMA Clinical Challenges via a Multi-Agent Conversational Framework

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Background & Objective Standard clinical LLM benchmarks use multiple-choice vignettes that present all information up front, unlike real encounters where clinicians iteratively elicit histories and objective data. We hypothesized that such formats inflate LLM performance and mask weaknesses in diagnostic reasoning. We developed and evaluated a multi-AI agent conversational framework that converts JAMA Clinical Challenge cases into multi-turn dialogues, and assessed its impact on diagnostic accuracy across frontier LLMs. Methods We adapted 815 diagnostic cases from 1,519 JAMA Clinical Challenges into two formats: (1) original vignette and (2) multi-agent conversation with a Patient AI (subjective history) and a System AI (objective data: exam, labs, imaging). A Clinical LLM queried these agents and produced a final diagnosis. Models tested were O1 (OpenAI), GPT-4o (OpenAI), LLaMA-3-70B (Meta), and Deepseek-R1-distill-LLaMA3-70B (Deepseek), each in multiple-choice and free-response modes. Free-response grading used a separate GPT-4o judge for diagnostic equivalence. Accuracy (Wilson 95% CIs) and conversation lengths were compared using two-tailed tests. Results Accuracy decreased for all models when moving from vignettes to conversations and from multiple-choice to free-response (p<0.0001 for all pairwise comparisons). In vignette multiple-choice, accuracy was O1 79.8% (95% CI, 76.9%–82.4%), GPT-4o 74.5% (71.4%–77.4%), LLaMA-3 70.9% (69.5%–72.2%), Deepseek-R1 69.0% (67.5%–70.4%). In conversation multiple-choice: O1 69.1% (65.8%–72.2%), GPT-4o 51.3% (49.8%–52.8%), LLaMA-3 49.7% (48.2%–51.3%), Deepseek-R1 34.0% (32.6%–35.5%). In conversation free-response: O1 31.7% (28.6%–34.9%), GPT-4o 20.7% (19.5%–22.0%), LLaMA-3 22.9% (21.6%–24.2%), Deepseek-R1 9.3% (8.4%–10.2%). O1 generally required fewer conversational turns than GPT-4o, suggesting more efficient multi-turn reasoning. Conclusions Converting vignettes into multi-agent, multi-turn dialogues reveals substantial performance drops across leading LLMs, indicating that static multiple-choice benchmarks overestimate clinical reasoning competence. Our open-source framework offers a more rigorous and discriminative evaluation and a realistic substrate for educational use, enabling assessment of iterative information-gathering and synthesis that better reflects clinical practice.
Full text 21,150 characters · extracted from preprint-html · click to expand
Evaluating Large Language Model Diagnostic Performance on JAMA Clinical Challenges via a Multi-Agent Conversational Framework | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Evaluating Large Language Model Diagnostic Performance on JAMA Clinical Challenges via a Multi-Agent Conversational Framework View ORCID Profile Karl L. Sangwon , Jeff Zhang , Robert Steele , Jaden Stryker , Jin Vivian Lee , Joanne Choi , Krithik Vishwanath , Daniel Alexander Alber , Douglas Kondziolka , Michal Mankowski , Eric Karl Oermann doi: https://doi.org/10.1101/2025.08.20.25334087 Karl L. Sangwon 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA B.S. Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Karl L. Sangwon For correspondence: Karl.Sangwon{at}nyulangone.org Jeff Zhang 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA M.S. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Robert Steele 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA M.S. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jaden Stryker 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA B.S. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jin Vivian Lee 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Joanne Choi 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA B.A. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Krithik Vishwanath 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA B.S. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Daniel Alexander Alber 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA B.S. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Douglas Kondziolka 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Michal Mankowski 4 Department of Surgery, NYU Langone Health , New York, NY 10016, USA Ph.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site Eric Karl Oermann 1 Department of Neurological Surgery, NYU Langone Health , New York, NY 10016, USA 2 Department of Radiology, NYU Langone Health , New York, NY 10016, USA 3 Center for Data Science, New York University , New York, NY 10011, USA 5 Neuroscience Institute, NYU Langone Health , New York, NY 10016, USA M.D. Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: Eric.Oermann{at}nyulangone.org Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Background & Objective Standard clinical LLM benchmarks use multiple-choice vignettes that present all information up front, unlike real encounters where clinicians iteratively elicit histories and objective data. We hypothesized that such formats inflate LLM performance and mask weaknesses in diagnostic reasoning. We developed and evaluated a multi-AI agent conversational framework that converts JAMA Clinical Challenge cases into multi-turn dialogues, and assessed its impact on diagnostic accuracy across frontier LLMs. Methods We adapted 815 diagnostic cases from 1,519 JAMA Clinical Challenges into two formats: (1) original vignette and (2) multi-agent conversation with a Patient AI (subjective history) and a System AI (objective data: exam, labs, imaging). A Clinical LLM queried these agents and produced a final diagnosis. Models tested were O1 (OpenAI), GPT-4o (OpenAI), LLaMA-3-70B (Meta), and Deepseek-R1-distill-LLaMA3-70B (Deepseek), each in multiple-choice and free-response modes. Free-response grading used a separate GPT-4o judge for diagnostic equivalence. Accuracy (Wilson 95% CIs) and conversation lengths were compared using two-tailed tests. Results Accuracy decreased for all models when moving from vignettes to conversations and from multiple-choice to free-response (p<0.0001 for all pairwise comparisons). In vignette multiple-choice, accuracy was O1 79.8% (95% CI, 76.9%–82.4%), GPT-4o 74.5% (71.4%–77.4%), LLaMA-3 70.9% (69.5%–72.2%), Deepseek-R1 69.0% (67.5%–70.4%). In conversation multiple-choice: O1 69.1% (65.8%–72.2%), GPT-4o 51.3% (49.8%–52.8%), LLaMA-3 49.7% (48.2%–51.3%), Deepseek-R1 34.0% (32.6%–35.5%). In conversation free-response: O1 31.7% (28.6%–34.9%), GPT-4o 20.7% (19.5%–22.0%), LLaMA-3 22.9% (21.6%–24.2%), Deepseek-R1 9.3% (8.4%–10.2%). O1 generally required fewer conversational turns than GPT-4o, suggesting more efficient multi-turn reasoning. Conclusions Converting vignettes into multi-agent, multi-turn dialogues reveals substantial performance drops across leading LLMs, indicating that static multiple-choice benchmarks overestimate clinical reasoning competence. Our open-source framework offers a more rigorous and discriminative evaluation and a realistic substrate for educational use, enabling assessment of iterative information-gathering and synthesis that better reflects clinical practice. Standard clinical LLM benchmarks rely on multiple-choice vignettes where all relevant information is presented upfront—failing to reflect how clinicians iteratively gather and synthesize data in real encounters. 1 - 4 We hypothesized that this format may inflate LLM performance and mask weaknesses in diagnostic reasoning. To address this, we developed and evaluated a multi-AI agent conversational framework that converts JAMA Clinical Challenge case vignettes into dynamic multi-turn dialogues simulating a more realistic diagnostic process. We assessed the effect of this conversational format on diagnostic accuracy across frontier LLMs. Methods We adapted 815 diagnostic cases from 1,519 publicly available JAMA Clinical Challenges into two formats: (1) original vignette and (2) multi-agent conversation involving Patient AI (subjective history) and System AI (objective clinical findings: exam, labs, imaging). A Clinical LLM interacts with these agents and synthesizes information to generate a diagnosis ( Figure 1 ). Prompt details are available in eTable ( Supplement 1) . Download figure Open in new tab Figure 1. Multi-AI Agent Framework for Converting Clinical Vignette Questions to Conversations. The framework transforms JAMA Clinical Challenge questions into dynamic conversations by separating information into subjective (patient-reported) and objective (clinical data) components. A Clinical AI or doctor navigates two distinct conversational interactions: (1) Patient-Doctor Conversation through a Patient AI that provides subjective information based on the patient’s description, and (2) System-Doctor Conversation through a Health System AI that supplies objective clinical data (laboratory results, vital signs, examination findings, and imaging). The Clinical AI or human user integrates information from both channels to formulate a final diagnosis. Example conversation snippets demonstrate the natural flow of history-taking and clinical data gathering through these parallel interfaces. We evaluated GPT-4o (OpenAI), LLaMA-3-70B (Meta), Deepseek-R1-distill-LLaMA3-70B (Deepseek), and O1 (OpenAI). Deepseek and O1 models were selected for their reported multi-step reasoning capabilities—Deepseek via training on mathematical problem-solving tasks, and O1 as a reasoning-optimized foundational model. Each model and case were tested in both multiple-choice and free-response formats. A separate GPT-4o model judged free-response answers for diagnostic equivalence. Conversations were checked for clinical coherence by medical trainees and attending physicians. Performance metrics included diagnostic accuracy and average conversation length. Accuracy was reported with 95% confidence intervals using Wilson’s methods and compared using 2-tailed t tests. Results O1 achieved the highest accuracy across all formats, followed by GPT4o, LlaMA-3 and Deepseek-R1. All models had lower accuracy in conversational formats vs vignettes, and in free-response vs multiple-choice formats (p<.0001, all pairwise). ( Figure 2A ) Download figure Open in new tab Figure 2. LLM Performance Analysis Across Assessment Formats. (A) Diagnostic accuracy of four different LLMs on JAMA Clinical Challenge questions across different formats. Performance dropped significantly when multiple choices were removed, and when converted into conversation format where models had to interact with other AI models to iteratively collect information for diagnosis (p<0.0001 for each). Error bars represent 95% confidence intervals. (B) Analysis of LLM conversation lengths across question formats. The average number of conversational exchanges for multiple choice was lower than free response formats. Error bars represent one standard deviation. **** denotes p<0.0001, *** denotes p<0.001, ** denotes p<0.01, * denotes p<0.05, ns denotes non-significant p-value. In vignette multiple-choice, accuracy was: O1, 79.8% (95%CI, 76.9%-82.4%); GPT-4o, 74.5% (71.4%-77.4%); LLaMA-3, 70.9% (69.5%-72.2%); Deepseek-R1, 69.0% (67.5%-70.4%). In vignette free-response: O1, 49.9% (46.5%-53.4%); GPT-4o, 40.1% (38.6%-41.6%); LLaMA-3, 41.4% (32.6%-35.6%); Deepseek-R1, 34.1% (39.9%-42.9%) (p<.0001, all pairwise). In conversation multiple-choice, O1 again led at 69.1% (65.8%-72.2%), followed by GPT-4o (51.3%, 49.8%-52.8%), LLaMA-3 (49.7%, 48.2%-51.3%), and Deepseek-R1 (34.0%, 32.6%-35.5%). In conversation free-response: O1, 31.7% (28.6%-34.9%); GPT-4o, 20.7% (19.5%-22.0%); LLaMA-3, 22.9% (21.6%-24.2%); Deepseek-R1, 9.3% (8.4%-10.2%) (p<.0001, all pairwise). In conversation multiple-choice, GPT-4o had the longest interactions (17.7 messages), followed by O1 (17.1), LLaMA-3 (16.2), and Deepseek-R1 (4.8). In conversation free-response, GPT-4o again led (20.9), followed by LLaMA-3 (19.4), O1 (14.3), and Deepseek-R1 (6.5) ( Figure 2B ). Discussion Our study suggests that static clinical vignettes overestimate LLM’s diagnostic reasoning capabilities. By simulating more realistic clinical encounters through a multi-agent, multi-turn conversational framework applied to JAMA Clinical Challenge cases, we observed substantial performance drops across models—particularly in free-response formats—underscoring challenges in iterative information gathering and synthesis. Deepseek-R1-distilled-Llama-3, a model trained for mathematical problem-solving reasoning, performed worst in free-response conversations—often failing to ask appropriate follow-up questions or retrieve key clinical details. This suggests that pretraining optimized for structured, single-turn tasks may limit multi-turn diagnostic reasoning. 4 In contrast, O1 consistently outperformed others across all formats and required fewer turns, suggesting more efficient and conversational reasoning. These findings indicate that traditional multiple-choice benchmarks may overstate LLMs’ clinical competence. 5 , 6 Conversational evaluations offer a more rigorous and discriminative approach, capable of distinguishing between models with and without advanced multi-turn reasoning capabilities. Our open-source framework provides a scalable, reproducible method for evaluating LLMs in more clinically reflective settings. Limitations include reliance on static vignettes as the information source and the absence of human comparator data. Future studies would benefit from expanded source data and evaluation of clinician-AI interactions using this framework for educational and diagnostic applications, particularly in domains requiring iterative clinical reasoning such as differential diagnosis or triage. Data Availability All data produced are available online at https://jamanetwork.com/collections/44038/clinical-challenge https://jamanetwork.com/collections/44038/clinical-challenge Acknowledgements Footnotes Conflict of Interest: None. Disclosures: EKO reports employment in Eikon Therapeutics; equity in Artisight Inc., Delvi Inc., MarchAI Inc.; consulting for Sofinnova, Google Disclosure of Funding: None. References 1. ↵ Harsha N , Nicholas K , McKinney SM , Dean C , Eric H. Capabilities of GPT-4 on medical challenge problems . arXiv [csCL]. Published online 2023 . 2. Strong E , DiGiammarino A , Weng Y , et al. Chatbot vs medical student performance on free-response clinical reasoning examinations . JAMA Intern Med . 2023 ; 183 ( 9 ): 1028 – 1030 . OpenUrl PubMed 3. Han T , Adams LC , Bressem KK , Busch F , Nebelung S , Truhn D. Comparative analysis of multimodal large language model performance on clinical vignette questions . JAMA . 2024 ; 331 ( 15 ): 1320 – 1321 . OpenUrl CrossRef PubMed 4. ↵ Laban P , Hayashi H , Zhou Y , Neville J. LLMs Get Lost In Multi-Turn Conversation . arXiv [csCL]. Published online 2025 . http://arxiv.org/abs/2505.06120 5. ↵ Alber DA , Yang Z , Alyakin A , et al. Medical large language models are vulnerable to data-poisoning attacks . Nat Med . 2025 ; 31 ( 2 ): 618 – 626 . doi: 10.1038/s41591-024-03445-1 OpenUrl CrossRef PubMed 6. ↵ Johri S , Jeong J , Tran BA , et al. An evaluation framework for clinical use of large language models in patient interaction tasks . Nat Med . 2025 ; 31 ( 1 ): 77 – 86 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted August 24, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Evaluating Large Language Model Diagnostic Performance on JAMA Clinical Challenges via a Multi-Agent Conversational Framework Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Evaluating Large Language Model Diagnostic Performance on JAMA Clinical Challenges via a Multi-Agent Conversational Framework Karl L. Sangwon , Jeff Zhang , Robert Steele , Jaden Stryker , Jin Vivian Lee , Joanne Choi , Krithik Vishwanath , Daniel Alexander Alber , Douglas Kondziolka , Michal Mankowski , Eric Karl Oermann medRxiv 2025.08.20.25334087; doi: https://doi.org/10.1101/2025.08.20.25334087 Share This Article: Copy Citation Tools Evaluating Large Language Model Diagnostic Performance on JAMA Clinical Challenges via a Multi-Agent Conversational Framework Karl L. Sangwon , Jeff Zhang , Robert Steele , Jaden Stryker , Jin Vivian Lee , Joanne Choi , Krithik Vishwanath , Daniel Alexander Alber , Douglas Kondziolka , Michal Mankowski , Eric Karl Oermann medRxiv 2025.08.20.25334087; doi: https://doi.org/10.1101/2025.08.20.25334087 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (297) Cardiovascular Medicine (4421) Dentistry and Oral Medicine (443) Dermatology (381) Emergency Medicine (606) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1507) Epidemiology (15212) Forensic Medicine (30) Gastroenterology (1121) Genetic and Genomic Medicine (6581) Geriatric Medicine (667) Health Economics (996) Health Informatics (4520) Health Policy (1366) Health Systems and Quality Improvement (1611) Hematology (539) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15906) Intensive Care and Critical Care Medicine (1103) Medical Education (620) Medical Ethics (144) Nephrology (667) Neurology (6580) Nursing (345) Nutrition (998) Obstetrics and Gynecology (1141) Occupational and Environmental Health (956) Oncology (3324) Ophthalmology (970) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (663) Pediatrics (1689) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5431) Public and Global Health (9212) Radiology and Imaging (2193) Rehabilitation Medicine and Physical Therapy (1368) Respiratory Medicine (1194) Rheumatology (593) Sexual and Reproductive Health (709) Sports Medicine (529) Surgery (709) Toxicology (99) Transplantation (288) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9ff1487ecda68e2e',t:'MTc3OTM0MjQxMg=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00