Full text
19,270 characters
· extracted from
preprint-html
· click to expand
A Unified Platform for Radiology Report Generation and Clinician-Centered AI Evaluation | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search A Unified Platform for Radiology Report Generation and Clinician-Centered AI Evaluation Zhuoqi Ma , Xinye Yang , Zach Atalay , Andrew Yang , Scott Collins , Harrison Bai , Michael Bernstein , Grayson Baird , Zhicheng Jiao doi: https://doi.org/10.1101/2025.07.07.25331018 Zhuoqi Ma 1 Radiology AI Lab, Brown University Health Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xinye Yang 2 Computer Science Department, Brown University Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zach Atalay 2 Computer Science Department, Brown University Find this author on Google Scholar Find this author on PubMed Search for this author on this site Andrew Yang 2 Computer Science Department, Brown University Find this author on Google Scholar Find this author on PubMed Search for this author on this site Scott Collins 1 Radiology AI Lab, Brown University Health Find this author on Google Scholar Find this author on PubMed Search for this author on this site Harrison Bai 3 Radiology Department, University of Colorado Find this author on Google Scholar Find this author on PubMed Search for this author on this site Michael Bernstein 1 Radiology AI Lab, Brown University Health Find this author on Google Scholar Find this author on PubMed Search for this author on this site Grayson Baird 1 Radiology AI Lab, Brown University Health Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zhicheng Jiao 1 Radiology AI Lab, Brown University Health Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Data/Code Preview PDF ABSTRACT Generative AI models have demonstrated strong potential in radiology report generation, but their clinical adoption depends on physician trust. In this study, we conducted a radiology-focused Turing test to evaluate how well attendings and residents distinguish AI-generated reports from those written by radiologists, and how their confidence and decision time reflect trust. we developed an integrated web-based platform comprising two core modules: Report Generation and Report Evaluation. Using the web-based platform, eight participants evaluated 48 anonymized X-ray cases, each paired with two reports from three comparison groups: radiologist vs. AI model 1, radiologist vs. AI model 2, and AI model 1 vs. AI model 2. Participants selected the AI-generated report, rated their confidence, and indicated report preference. Attendings outperformed residents in identifying AI-generated reports (49.9% vs. 41.1%) and exhibited longer decision times, suggesting more deliberate judgment. Both groups took more time when both reports were AI-generated. Our findings highlight the role of clinical experience in AI acceptance and the need for design strategies that foster trust in clinical applications. The project page of the evaluation platform is available at: https://zachatalay89.github.io/Labsite . 1 Background In recent years, the rapid advancement of generative AI technologies has significantly propelled the development of the medical and healthcare domains. Among them, large language models have demonstrated remarkable capabilities in language generation and question answering across various tasks, particularly in radiology report generation for X-rays, CT scans, and even MRIs[ 1 ]. Studies have shown that generative AI models can achieve notable performance, even reaching accuracy and clinical value comparable to those of radiologists [ 2 ]. Despite the promising capabilities of generative AI in radiology report generation, successful integration into clinical workflows hinges on physicians’ acceptance and trust [ 3 , 4 ]. This is primarily due to the high-stakes nature of medical decision-making, where even occasional errors can have serious consequences. Physicians must not only be confident in the accuracy of AI-generated reports, but also trust their consistency, reliability, and alignment with clinical standards.As a result, beyond technical performance, earning the trust of end-users is a critical prerequisite for the widespread adoption of AI-generated reports in real-world clinical settings. In this study, we conducted a radiology-oriented Turing test to investigate how well physicians and residents distinguish between AI-generated reports and reports generated by radiologists; and how physicians and residents’ knowledge influence their judgement time and confidence. 2 Platform Design To facilitate both the generation and evaluation of AI-generated radiology reports, we developed an integrated web-based platform comprising two core modules: Report Generation and Report Evaluation, as shown in Figure 1 . Download figure Open in new tab Figure 1: Overview of the radiology report generation and evaluation framework. 2.1 Report Generation This module allows users to upload medical images in various formats (e.g., DICOM, JPG, PNG) via drag-and-drop or file selection. Upon upload, the system automatically processes the image and generates a radiology report using state- of-the-art AI models. The generated report is displayed in a structured format, including metadata (e.g., examination type, technique), findings, and impression. Users can directly edit, copy, or reset the generated report within the interface, enabling efficient post-editing and clinical review. 2.2 Report Evaluation To assess the quality and credibility of AI-generated reports, we designed a Turing test-style evaluation module. Users are presented with two radiology reports side by side, along with the corresponding medical image and an interactive image viewer that supports pan, window/level adjustment, and rotation. Participants are asked to identify which report (if any) they believe is AI-generated, followed by a confidence rating. This enables both qualitative and quantitative assessment of report realism and trustworthiness across different user groups. By integrating generation and evaluation functionalities within a unified interface, our platform provides a complete end-to-end pipeline for studying human perception of AI-generated medical content, while also supporting clinical usability testing and iterative model improvement. 3 Evaluation Pipeline We use the web-based radiology report evaluation platform to assess physicians’ trust and acceptance of AI-generated radiology reports through a Turing-test-inspired interface. In report evaluation platform, the interface presents a medical image (e.g., an X-ray or CT slice) at the top, accompanied by two diagnostic reports labeled Report 1 and Report 2. Each report contains detailed findings describing various anatomical and pathological observations. The physician is asked to review both reports and the corresponding image, then respond to three key questions: (1) Preference – Which report would you prefer to use in clinical practice? (2) AI Identification – Which report do you believe was generated by AI? (3) Confidence – How confident are you in your previous judgment? To ensure fair evaluation, report identities (AI vs. human) are randomly shuffled across cases. Also, report formatting is standardized to reduce stylistic cues. No model or author identifiers are exposed during the task. 48 X-ray cases and radiologists’ reports were retrieved from Rhode Island Hospital. All data were collected anonymously to ensure participant privacy. For each X-ray case, we generate AI reports using two state-of-the-art methods[ 5 , 6 ]. The cases were evenly divided into three comparison groups based on report generation methods: radiologist vs. AI model 1, radiologist vs. AI model 2, and AI model 1 vs. AI model 2 (16 each condition). Eight participants (4 attendings and 4 residents) participated in this experiment. After reading the provided case, participants were asked to identify which report was AI-generated. Additionally, they provided Likert scale evaluations on a scale of 1(low confidence) to 5(high confidence) on their confidence in these judgements. Finally, participants indicate which report they would be more inclined to adopt in clinical practice. This study was IRB approved. 4 Results In Table 1 , we report the accuracy, time and confidence between residents and attendings. Attendings were correct 49.9% and residents 41.1% of the time in identifying AI-generated reports, both higher than chance (33.3%), indicating that radiologists largely beat the Turing test and more so than residents. Attendings were more likely to distinguish AI-generated reports from radiologist-generated reports than residents, and both groups were able to distinguish over chance. This suggests that clinical experience helps mitigate bias toward AI-generated content, whereas residents are more prone to perception-driven preference shifts. Attendings exhibited longer average response times (56.84 seconds) compared to residents (31.86 seconds), suggesting a more analytical decision-making approach, p<.01. Furthermore, both groups demonstrated increased response times (attendings by 18.87% and residents by 29%) when evaluating cases in which both reports were AI-generated, indicating that the presence of fully synthetic content introduces greater cognitive load and uncertainty, prompting more prolonged evaluation. Attendings showed slightly more confidence than residents, but this was not significant. For both groups, for every one-unit increase in confidence, the odds of being correct increased 20%, p=.025. View this table: View inline View popup Download powerpoint Table 1: Performance Comparison Between Residents and Attendings 5 Conclusion Our findings reveal important differences between attendings and residents in terms of acceptance and bias toward AI-generated reports, highlighting the need for targeted strategies to foster trust in AI tools and ensure their safe and effective integration into clinical workflows. Data Availability All data produced in the present study are available upon reasonable request to the authors https://zachatalay89.github.io/Labsite/ Footnotes author{at}example.com References [1]. ↵ Cheng-Yi Li , Kao-Jung Chang , Cheng-Fu Yang , Hsin-Yu Wu , Wenting Chen , Hritik Bansal , Ling Chen , Yi-Ping Yang , Yu-Chun Chen , Shih-Pin Chen , et al. Towards a holistic framework for multimodal llm in 3d brain ct radiology report generation . Nature Communications , 16 ( 1 ): 2258 , 2025 . OpenUrl PubMed [2]. ↵ Eun Kyoung Hong , Byungseok Roh , Beomhee Park , Jae-Bock Jo , Woong Bae , Jai Soung Park , and Dong-Wook Sung . Value of using a generative ai model in chest radiography reporting: a reader study . Radiology , 314 ( 3 ): e241646 , 2025 . OpenUrl PubMed [3]. ↵ Ryutaro Tanno , David GT Barrett , Andrew Sellergren , Sumedh Ghaisas , Sumanth Dathathri , Abigail See , Johannes Welbl , Charles Lau , Tao Tu , Shekoofeh Azizi , et al. Collaboration between clinicians and vision–language models in radiology report generation . Nature Medicine , 31 ( 2 ): 599 – 608 , 2025 . OpenUrl CrossRef PubMed [4]. ↵ Shruthi Shekar , Pat Pataranutaporn , Chethan Sarabu , Guillermo A Cecchi , and Pattie Maes . People overtrust ai-generated medical advice despite low accuracy . NEJM AI , page AIoa2300015 , 2025 . [5]. ↵ Kang Liu , Zhuoqi Ma , Xiaolu Kang , Zhusi Zhong , Zhicheng Jiao , Grayson Baird , Harrison Bai , and Qiguang Miao . Structural entities extraction and patient indications incorporation for chest x-ray report generation . In International Conference on Medical Image Computing and Computer-Assisted Intervention , pages 433 – 443 . Springer , 2024 . [6]. ↵ Kang Liu , Zhuoqi Ma , Xiaolu Kang , Yunan Li , Kun Xie , Zhicheng Jiao , and Qiguang Miao . Enhanced contrastive learning with multi-view longitudinal data for chest x-ray report generation . arXiv preprint arXiv: 2502.20056 , 2025 . View the discussion thread. Back to top Previous Next Posted July 08, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A Unified Platform for Radiology Report Generation and Clinician-Centered AI Evaluation Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A Unified Platform for Radiology Report Generation and Clinician-Centered AI Evaluation Zhuoqi Ma , Xinye Yang , Zach Atalay , Andrew Yang , Scott Collins , Harrison Bai , Michael Bernstein , Grayson Baird , Zhicheng Jiao medRxiv 2025.07.07.25331018; doi: https://doi.org/10.1101/2025.07.07.25331018 Share This Article: Copy Citation Tools A Unified Platform for Radiology Report Generation and Clinician-Centered AI Evaluation Zhuoqi Ma , Xinye Yang , Zach Atalay , Andrew Yang , Scott Collins , Harrison Bai , Michael Bernstein , Grayson Baird , Zhicheng Jiao medRxiv 2025.07.07.25331018; doi: https://doi.org/10.1101/2025.07.07.25331018 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4436) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4538) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (542) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3333) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00e8775db48fff4',t:'MTc3OTY0OTA3OA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.